Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
25 commits
Select commit Hold shift + click to select a range
abf3355
store bitpacked block offsets as an optional child
mhk197 Oct 2, 2026
477bfb9
check bitpacked widths once per kernel
mhk197 Oct 2, 2026
5f3f2c3
only check bitpacked block boundaries in debug builds
mhk197 Oct 2, 2026
cf2c375
move bitpack_compress into a directory
mhk197 Oct 2, 2026
861873b
encode bitpacked arrays with a width per block
mhk197 Oct 2, 2026
1729b6d
rename bitpacked encode benches
mhk197 Oct 2, 2026
b7bcace
only zero-pad a partial last block when bitpacking
mhk197 Oct 2, 2026
b369fb6
keep the global bitpack encoder unchanged
mhk197 Oct 2, 2026
8c525af
move global-only bitpacking code back into global.rs
mhk197 Oct 2, 2026
a109b8c
fix clippy
mhk197 Oct 2, 2026
95f5480
bench only the blocked packing kernel
mhk197 Oct 2, 2026
de28cfc
fix comment
mhk197 Oct 2, 2026
0d407fd
move bitpack_decompress into a directory
mhk197 Oct 2, 2026
385b15a
decode bitpacked blocks with variable bit widths
mhk197 Oct 2, 2026
c8d26f8
trim blocked decode bench to one case per type
mhk197 Oct 2, 2026
67da367
round-trip blocked bitpacking and encode the decode bench input
mhk197 Oct 2, 2026
16b8075
dispatch bitpacked decode in callers and drop blocked mapped decode
mhk197 Oct 2, 2026
94c7da6
serialize blocked bitpacked arrays as fastlanes.bitpacked.v2
mhk197 Oct 2, 2026
ca053fd
fix import order
mhk197 Oct 2, 2026
c61995a
refine BitPackingScheme to per-block bit widths when fastlanes.bitpac…
mhk197 Oct 2, 2026
336bbfc
fix clippy
mhk197 Oct 2, 2026
a500896
Merge branch 'mk/bitpacked-v2-wire' into mk/bitpacked-v2-scheme
mhk197 Oct 2, 2026
9b4777b
always produce blocked arrays in bitpacking v2
mhk197 Oct 2, 2026
c98cb99
keep the dictionary test out of blocked bitpacking's reach
mhk197 Oct 2, 2026
ddc804c
slice blocked bitpacked arrays without decoding
mhk197 Oct 2, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions encodings/fastlanes/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,10 @@ _test-harness = ["dep:rand"]
name = "bitpacking_take"
harness = false

[[bench]]
name = "bitpack_blocked_decompress"
harness = false

[[bench]]
name = "canonicalize_bench"
harness = false
Expand All @@ -69,6 +73,10 @@ harness = false
name = "bitpack_compare_sweep"
harness = false

[[bench]]
name = "bitpack_blocked"
harness = false

[[bench]]
name = "cast_bitpacked"
harness = false
Expand Down
52 changes: 52 additions & 0 deletions encodings/fastlanes/benches/bitpack_blocked.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
// SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright the Vortex contributors

//! Benchmarks `bitpack_encode_blocked`, which packs every 1024-value block at its own bit width.

#![expect(clippy::unwrap_used)]

use std::sync::LazyLock;

use divan::Bencher;
use divan::counter::ItemsCount;
use mimalloc::MiMalloc;
use vortex_array::VortexSessionExecute;
use vortex_array::arrays::PrimitiveArray;
use vortex_fastlanes::bitpack_compress::bitpack_encode_blocked;
use vortex_session::VortexSession;

#[global_allocator]
static GLOBAL: MiMalloc = MiMalloc;

fn main() {
divan::main();
}

static SESSION: LazyLock<VortexSession> = LazyLock::new(|| {
let session = vortex_array::array_session();
vortex_fastlanes::initialize(&session);
session
});

/// 64 blocks of `u32`, 256 KiB in all: shorter iterations are noisy on the walltime legs.
const NUM_BLOCKS: usize = 64;

#[vortex_bench_support::cpu_features]
#[divan::bench]
fn bitpack_blocked_compress(bencher: Bencher) {
// Block widths cycle from 1 to 16 bits, so the blocks take different packing kernels.
let bit_widths: Vec<u8> = (1..=16).cycle().take(NUM_BLOCKS).collect();
let array = PrimitiveArray::from_iter(bit_widths.iter().flat_map(|&bit_width| {
(0..1024u32).map(move |i| i.wrapping_mul(7919) & ((1 << bit_width) - 1))
}));

bencher.counter(ItemsCount::new(array.len())).bench(|| {
bitpack_encode_blocked(
&array,
&bit_widths,
Some(0),
&mut SESSION.create_execution_ctx(),
)
.unwrap()
});
}
69 changes: 69 additions & 0 deletions encodings/fastlanes/benches/bitpack_blocked_decompress.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
// SPDX-License-Identifier: Apache-2.0
// SPDX-FileCopyrightText: Copyright the Vortex contributors

//! Benchmarks decoding a `BitPacked` array whose 1024-value blocks each have their own bit width.

#![expect(clippy::unwrap_used)]

use std::sync::LazyLock;

use divan::Bencher;
use divan::counter::ItemsCount;
use mimalloc::MiMalloc;
use num_traits::AsPrimitive;
use vortex_array::VortexSessionExecute;
use vortex_array::arrays::PrimitiveArray;
use vortex_array::dtype::NativePType;
use vortex_fastlanes::BitPackedArraySlotsExt;
use vortex_fastlanes::bitpack_compress::bitpack_encode_blocked;
use vortex_fastlanes::bitpack_decompress::unpack_array_blocked;
use vortex_session::VortexSession;

#[global_allocator]
static GLOBAL: MiMalloc = MiMalloc;

fn main() {
divan::main();
}

static SESSION: LazyLock<VortexSession> = LazyLock::new(|| {
let session = vortex_array::array_session();
vortex_fastlanes::initialize(&session);
session
});

/// 64 blocks per array: shorter iterations are noisy on the walltime legs.
const NUM_BLOCKS: usize = 64;

#[vortex_bench_support::cpu_features]
#[divan::bench(types = [u32, u64])]
fn bitpack_blocked_decompress<T>(bencher: Bencher)
where
T: NativePType,
u64: AsPrimitive<T>,
{
// Block widths cycle from 1 to 16 bits, so the blocks take different unpacking kernels.
let bit_widths: Vec<u8> = (1..=16).cycle().take(NUM_BLOCKS).collect();
let values = PrimitiveArray::from_iter(bit_widths.iter().flat_map(|&bit_width| {
(0..1024u64).map(move |i| {
AsPrimitive::<T>::as_(i.wrapping_mul(7919) & ((1 << bit_width) - 1))
})
}));
let array = bitpack_encode_blocked(
&values,
&bit_widths,
Some(0),
&mut SESSION.create_execution_ctx(),
)
.unwrap();
let offsets = array.block_offsets().unwrap().clone();

bencher.counter(ItemsCount::new(array.len())).bench(|| {
unpack_array_blocked(
array.as_view(),
&offsets,
&mut SESSION.create_execution_ctx(),
)
.unwrap()
});
}
6 changes: 5 additions & 1 deletion encodings/fastlanes/benches/bitpack_compare.rs
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ use vortex_buffer::BufferMut;
use vortex_fastlanes::BitPacked;
use vortex_fastlanes::BitPackedArray;
use vortex_fastlanes::BitPackedData;
use vortex_fastlanes::BitWidths;
use vortex_session::VortexSession;

#[global_allocator]
Expand Down Expand Up @@ -58,12 +59,15 @@ const BIT_WIDTHS: &[u8] = &[4, 16];
fn page_aligned(array: BitPackedArray) -> BitPackedArray {
let ptype = array.dtype().as_ptype();
let parts = BitPacked::into_parts(array);
let BitWidths::Global(bit_width) = parts.bit_widths else {
unreachable!("bitpack_encode packs every block at one bit width")
};
BitPacked::try_new(
parts.packed.ensure_aligned(Alignment::new(4096)).unwrap(),
ptype,
parts.validity,
parts.patches,
parts.bit_width,
bit_width,
parts.len,
parts.offset,
)
Expand Down
6 changes: 5 additions & 1 deletion encodings/fastlanes/benches/bitpack_compare_sweep.rs
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,7 @@ use vortex_buffer::BufferMut;
use vortex_fastlanes::BitPacked;
use vortex_fastlanes::BitPackedArray;
use vortex_fastlanes::BitPackedData;
use vortex_fastlanes::BitWidths;
use vortex_session::VortexSession;

#[global_allocator]
Expand Down Expand Up @@ -84,12 +85,15 @@ impl_bench_int!(u8, u16, u32, u64, i8, i16, i32, i64);
fn page_aligned(array: BitPackedArray) -> BitPackedArray {
let ptype = array.dtype().as_ptype();
let parts = BitPacked::into_parts(array);
let BitWidths::Global(bit_width) = parts.bit_widths else {
unreachable!("bitpack_encode packs every block at one bit width")
};
BitPacked::try_new(
parts.packed.ensure_aligned(Alignment::new(4096)).unwrap(),
ptype,
parts.validity,
parts.patches,
parts.bit_width,
bit_width,
parts.len,
parts.offset,
)
Expand Down
Loading
Loading