Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

45 changes: 25 additions & 20 deletions vortex-array/benches/compare.rs
Original file line number Diff line number Diff line change
Expand Up @@ -130,44 +130,49 @@ fn compare_bool_constant(bencher: Bencher) {
bench_compare(bencher, arr, constant, Operator::Eq);
}

#[divan::bench]
fn compare_int(bencher: Bencher) {
/// Every comparison operator, because the primitive kernel dispatches on the operator outside
/// the lane loop: each one is a separate instantiation that vectorizes on its own terms, and
/// measuring `Gte` alone said nothing about the other five.
const COMPARE_OPERATORS: &[Operator] = &[
Operator::Eq,
Operator::NotEq,
Operator::Gt,
Operator::Gte,
Operator::Lt,
Operator::Lte,
];

#[divan::bench(args = COMPARE_OPERATORS)]
fn compare_int(bencher: Bencher, op: Operator) {
let mut rng = StdRng::seed_from_u64(0);
let arr1 = int_array(&mut rng);
let arr2 = int_array(&mut rng);
bench_compare(bencher, arr1, arr2, Operator::Gte);
bench_compare(bencher, arr1, arr2, op);
}

#[divan::bench]
fn compare_int_nullable(bencher: Bencher) {
#[divan::bench(args = COMPARE_OPERATORS)]
fn compare_int_nullable(bencher: Bencher, op: Operator) {
let mut rng = StdRng::seed_from_u64(0);
let arr1 = int_array_nullable(&mut rng);
let arr2 = int_array_nullable(&mut rng);
bench_compare(bencher, arr1, arr2, Operator::Gte);
bench_compare(bencher, arr1, arr2, op);
}

#[divan::bench]
fn compare_int_constant(bencher: Bencher) {
#[divan::bench(args = COMPARE_OPERATORS)]
fn compare_int_constant(bencher: Bencher, op: Operator) {
let mut rng = StdRng::seed_from_u64(0);
let arr = int_array(&mut rng);
let constant = ConstantArray::new(50_000_000i64, ARRAY_SIZE).into_array();
bench_compare(bencher, arr, constant, Operator::Gte);
}

#[divan::bench]
fn compare_int_eq(bencher: Bencher) {
let mut rng = StdRng::seed_from_u64(0);
let arr1 = int_array(&mut rng);
let arr2 = int_array(&mut rng);
bench_compare(bencher, arr1, arr2, Operator::Eq);
bench_compare(bencher, arr, constant, op);
}

#[divan::bench]
fn compare_float(bencher: Bencher) {
/// Float lanes carry Vortex's total ordering, so the predicate is more than a machine compare.
#[divan::bench(args = COMPARE_OPERATORS)]
fn compare_float(bencher: Bencher, op: Operator) {
let mut rng = StdRng::seed_from_u64(0);
let arr1 = float_array(&mut rng);
let arr2 = float_array(&mut rng);
bench_compare(bencher, arr1, arr2, Operator::Gte);
bench_compare(bencher, arr1, arr2, op);
}

#[divan::bench]
Expand Down
38 changes: 34 additions & 4 deletions vortex-buffer/benches/collect_bool.rs
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,13 @@
//! through `cfg(target_feature)`, and how well the scalar loop auto-vectorizes depends on
//! the build. Comparing them across legs is the point.
//!
//! The public entry points are measured twice, under two names. `collect_bool_u32_gt` and
//! `from_bool_slice` stay untagged and keep their simulation history; the `*_dispatch` siblings
//! carry `#[cpu_features]` and measure the same call on the walltime legs, where predicate
//! evaluation and allocation sit alongside the pack loop as they do in shipped code. Splitting
//! them rather than tagging the originals is what keeps both modes: a tagged benchmark is
//! skipped in simulation, so one name cannot serve both.
//!
//! The hand-written per-kernel benchmarks are not tagged. Each one needs an instruction set
//! extension the other legs do not build for, so they stay out of CodSpeed entirely and
//! remain local A/B tools, as do the historical bit-at-a-time baselines.
Expand Down Expand Up @@ -194,12 +201,23 @@ fn from_bool_slice_old_scalar(bencher: Bencher, len: usize) {
.bench_refs(|words| collect_bool_words_old(words, len, |i| bools[i]));
}

#[divan::bench(args = INPUT_SIZE)]
fn from_bool_slice(bencher: Bencher, len: usize) {
/// The shipped `&[bool]` conversion: gather, multiversioned pack, and allocation together.
fn bench_from_bool_slice(bencher: Bencher, len: usize) {
let bools = make_bools(len);
bencher.bench(|| vortex_buffer::BitBufferMut::from(bools.as_slice()));
}

#[divan::bench(args = INPUT_SIZE)]
fn from_bool_slice(bencher: Bencher, len: usize) {
bench_from_bool_slice(bencher, len);
}

#[vortex_bench_support::cpu_features]
#[divan::bench(args = INPUT_SIZE)]
fn from_bool_slice_dispatch(bencher: Bencher, len: usize) {
bench_from_bool_slice(bencher, len);
}

#[cfg(not(codspeed))]
#[divan::bench(args = INPUT_SIZE)]
fn collect_bool_u32_gt_old_scalar(bencher: Bencher, len: usize) {
Expand All @@ -209,8 +227,20 @@ fn collect_bool_u32_gt_old_scalar(bencher: Bencher, len: usize) {
.bench_refs(|words| collect_bool_words_old(words, len, |i| values[i] > u32::MAX / 2));
}

#[divan::bench(args = INPUT_SIZE)]
fn collect_bool_u32_gt(bencher: Bencher, len: usize) {
/// The shipped `BitBuffer::collect_bool` entry point under a `u32` comparison predicate:
/// predicate evaluation, the multiversioned pack loop, and allocation together.
fn bench_collect_bool_u32_gt(bencher: Bencher, len: usize) {
let values = make_u32s(len);
bencher.bench(|| BitBuffer::collect_bool(len, |i| values[i] > u32::MAX / 2));
}

#[divan::bench(args = INPUT_SIZE)]
fn collect_bool_u32_gt(bencher: Bencher, len: usize) {
bench_collect_bool_u32_gt(bencher, len);
}

#[vortex_bench_support::cpu_features]
#[divan::bench(args = INPUT_SIZE)]
fn collect_bool_u32_gt_dispatch(bencher: Bencher, len: usize) {
bench_collect_bool_u32_gt(bencher, len);
}
5 changes: 5 additions & 0 deletions vortex-compute/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -28,10 +28,15 @@ arrow-schema = { workspace = true }
divan = { workspace = true }
num-traits = { workspace = true }
rand = { workspace = true }
vortex-bench-support = { workspace = true }

[lints]
workspace = true

[[bench]]
name = "lane_kernels"
harness = false

[[bench]]
name = "compare_bits"
harness = false
209 changes: 209 additions & 0 deletions vortex-compute/benches/compare_bits.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,209 @@
// SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright the Vortex contributors

//! Benchmarks for the lowest-level comparison kernels: bit-packing a comparison predicate
//! over lanes with [`IndexedSourceExt::map_bits_into`].
//!
//! This is the kernel every native Vortex comparison bottoms out in. `vortex-array`'s
//! `collect_zip_bits` / `collect_bits` — used by the primitive and decimal compare paths —
//! are the allocation plus a `map_bits_into` call, so the shapes here are:
//!
//! - `zip_bits_*` / `const_bits_*`: the kernel alone, writing into caller-owned words.
//! Array-vs-array via [`LaneZip`], and array-vs-constant over a plain slice.
//! - `collect_zip_bits_*` / `collect_bits_*`: the same kernels with the `BufferMut<u64>`
//! allocation and [`BitBuffer`] freeze around them, mirroring `vortex-array` exactly.
//!
//! The `zip_bits_*` and `const_bits_*` benchmarks carry `#[cpu_features]`, so they are
//! measured on every walltime CPU-feature leg instead of in simulation. They are written once
//! and compiled differently per leg: the kernel is a branch-free lane loop whose whole cost is
//! how well it auto-vectorizes for the build, which is what comparing legs measures. The
//! `collect_*` benchmarks are untagged and stay in the sharded simulation job, where the
//! allocate-and-freeze wrapper is the part worth watching for instruction-count regressions.
//!
//! Integer lanes compare with their natural ordering; float lanes use `f64::total_cmp`, which
//! is the total ordering `vortex-array`'s `NativePType::is_lt` and friends are built on. The
//! `i128` lanes stand in for decimal comparison, which uses the same kernel.

use divan::Bencher;
use rand::SeedableRng;
use rand::prelude::*;
use rand::rngs::StdRng;
use vortex_buffer::BitBuffer;
use vortex_buffer::Buffer;
use vortex_buffer::BufferMut;
use vortex_compute::lane_kernels::IndexedSourceExt;
use vortex_compute::lane_kernels::LaneZip;

fn main() {
divan::main();
}

/// Sized to keep CodSpeed simulation under 1ms per benchmark, matching
/// `vortex-array`'s `compare` benchmarks.
const SIZES: &[usize] = &[8_192];

/// Two operand buffers plus a scalar operand, all drawn from one seeded RNG.
struct Fixture<T> {
lhs: Buffer<T>,
rhs: Buffer<T>,
constant: T,
}

fn fixture<T, F>(n: usize, mut sample: F) -> Fixture<T>
where
F: FnMut(&mut StdRng) -> T,
{
let mut rng = StdRng::seed_from_u64(0xC0FFEE);
let lhs = (0..n).map(|_| sample(&mut rng)).collect::<Buffer<T>>();
let rhs = (0..n).map(|_| sample(&mut rng)).collect::<Buffer<T>>();
let constant = sample(&mut rng);
Fixture { lhs, rhs, constant }
}

fn i32_fixture(n: usize) -> Fixture<i32> {
fixture(n, |rng| rng.random_range(0i32..100_000_000))
}

fn i64_fixture(n: usize) -> Fixture<i64> {
fixture(n, |rng| rng.random_range(0i64..100_000_000))
}

fn f64_fixture(n: usize) -> Fixture<f64> {
fixture(n, |rng| rng.random_range(0.0f64..1.0))
}

fn i128_fixture(n: usize) -> Fixture<i128> {
fixture(n, |rng| rng.random_range(0i128..100_000_000))
}

fn words(n: usize) -> Vec<u64> {
vec![0u64; n.div_ceil(64)]
}

/// Bit-pack `f(lhs[i], rhs[i])` into caller-owned words — the array-vs-array kernel.
fn bench_zip<T: Copy + Sync>(
bencher: Bencher,
n: usize,
f: Fixture<T>,
predicate: impl Fn(T, T) -> bool + Sync,
) {
bencher.with_inputs(|| words(n)).bench_refs(|out| {
LaneZip::new(f.lhs.as_slice(), f.rhs.as_slice())
.map_bits_into(out.as_mut_slice(), |(a, b)| predicate(a, b));
});
}

/// Bit-pack `f(lhs[i], constant)` into caller-owned words — the array-vs-constant kernel.
fn bench_const<T: Copy + Sync>(
bencher: Bencher,
n: usize,
f: Fixture<T>,
predicate: impl Fn(T, T) -> bool + Sync,
) {
bencher.with_inputs(|| words(n)).bench_refs(|out| {
f.lhs
.as_slice()
.map_bits_into(out.as_mut_slice(), |a| predicate(a, f.constant));
});
}

/// `vortex-array`'s `collect_zip_bits`: allocate the words, run the kernel, freeze the bits.
fn collect_zip_bits<T: Copy>(lhs: &[T], rhs: &[T], predicate: impl Fn(T, T) -> bool) -> BitBuffer {
let len = lhs.len();
let mut words = BufferMut::<u64>::zeroed(len.div_ceil(64));
LaneZip::new(lhs, rhs).map_bits_into(words.as_mut_slice(), |(a, b)| predicate(a, b));
bit_buffer_from_words(words, len)
}

/// `vortex-array`'s `collect_bits`, the array-vs-constant counterpart.
fn collect_bits<T: Copy>(values: &[T], predicate: impl Fn(T) -> bool) -> BitBuffer {
let len = values.len();
let mut words = BufferMut::<u64>::zeroed(len.div_ceil(64));
values.map_bits_into(words.as_mut_slice(), predicate);
bit_buffer_from_words(words, len)
}

fn bit_buffer_from_words(words: BufferMut<u64>, len: usize) -> BitBuffer {
let mut bytes = words.into_byte_buffer();
bytes.truncate(len.div_ceil(8));
BitBuffer::new(bytes.freeze(), len)
}

// -----------------------------------------------------------------------------
// Kernel benchmarks, measured per CPU-feature leg.
// -----------------------------------------------------------------------------

#[vortex_bench_support::cpu_features]
#[divan::bench(args = SIZES)]
fn zip_bits_i32_gte(bencher: Bencher, n: usize) {
bench_zip(bencher, n, i32_fixture(n), |a, b| a >= b);
}

#[vortex_bench_support::cpu_features]
#[divan::bench(args = SIZES)]
fn zip_bits_i64_gte(bencher: Bencher, n: usize) {
bench_zip(bencher, n, i64_fixture(n), |a, b| a >= b);
}

#[vortex_bench_support::cpu_features]
#[divan::bench(args = SIZES)]
fn zip_bits_i64_eq(bencher: Bencher, n: usize) {
bench_zip(bencher, n, i64_fixture(n), |a, b| a == b);
}

/// Float lanes under Vortex's total ordering, which is what makes this kernel harder to
/// vectorize than the integer ones.
#[vortex_bench_support::cpu_features]
#[divan::bench(args = SIZES)]
fn zip_bits_f64_lt(bencher: Bencher, n: usize) {
bench_zip(bencher, n, f64_fixture(n), |a: f64, b: f64| {
a.total_cmp(&b).is_lt()
});
}

/// `i128` lanes: the decimal compare path runs the same kernel over double-width lanes.
#[vortex_bench_support::cpu_features]
#[divan::bench(args = SIZES)]
fn zip_bits_i128_gte(bencher: Bencher, n: usize) {
bench_zip(bencher, n, i128_fixture(n), |a, b| a >= b);
}

#[vortex_bench_support::cpu_features]
#[divan::bench(args = SIZES)]
fn const_bits_i64_gte(bencher: Bencher, n: usize) {
bench_const(bencher, n, i64_fixture(n), |a, b| a >= b);
}

#[vortex_bench_support::cpu_features]
#[divan::bench(args = SIZES)]
fn const_bits_f64_lt(bencher: Bencher, n: usize) {
bench_const(bencher, n, f64_fixture(n), |a: f64, b: f64| {
a.total_cmp(&b).is_lt()
});
}

// -----------------------------------------------------------------------------
// Allocate-and-freeze wrappers, measured in simulation.
// -----------------------------------------------------------------------------

#[divan::bench(args = SIZES)]
fn collect_zip_bits_i64_gte(bencher: Bencher, n: usize) {
let f = i64_fixture(n);
bencher.bench(|| collect_zip_bits(f.lhs.as_slice(), f.rhs.as_slice(), |a, b| a >= b));
}

#[divan::bench(args = SIZES)]
fn collect_zip_bits_f64_lt(bencher: Bencher, n: usize) {
let f = f64_fixture(n);
bencher.bench(|| {
collect_zip_bits(f.lhs.as_slice(), f.rhs.as_slice(), |a: f64, b: f64| {
a.total_cmp(&b).is_lt()
})
});
}

#[divan::bench(args = SIZES)]
fn collect_bits_i64_gte(bencher: Bencher, n: usize) {
let f = i64_fixture(n);
bencher.bench(|| collect_bits(f.lhs.as_slice(), |a| a >= f.constant));
}
Loading