diff --git a/encodings/fastlanes/Cargo.toml b/encodings/fastlanes/Cargo.toml index 84e1e811966..08d75d70004 100644 --- a/encodings/fastlanes/Cargo.toml +++ b/encodings/fastlanes/Cargo.toml @@ -51,6 +51,10 @@ _test-harness = ["dep:rand"] name = "bitpacking_take" harness = false +[[bench]] +name = "bitpack_blocked_decompress" +harness = false + [[bench]] name = "canonicalize_bench" harness = false @@ -69,6 +73,10 @@ harness = false name = "bitpack_compare_sweep" harness = false +[[bench]] +name = "bitpack_blocked" +harness = false + [[bench]] name = "cast_bitpacked" harness = false diff --git a/encodings/fastlanes/benches/bitpack_blocked.rs b/encodings/fastlanes/benches/bitpack_blocked.rs new file mode 100644 index 00000000000..2223628e787 --- /dev/null +++ b/encodings/fastlanes/benches/bitpack_blocked.rs @@ -0,0 +1,52 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +//! Benchmarks `bitpack_encode_blocked`, which packs every 1024-value block at its own bit width. + +#![expect(clippy::unwrap_used)] + +use std::sync::LazyLock; + +use divan::Bencher; +use divan::counter::ItemsCount; +use mimalloc::MiMalloc; +use vortex_array::VortexSessionExecute; +use vortex_array::arrays::PrimitiveArray; +use vortex_fastlanes::bitpack_compress::bitpack_encode_blocked; +use vortex_session::VortexSession; + +#[global_allocator] +static GLOBAL: MiMalloc = MiMalloc; + +fn main() { + divan::main(); +} + +static SESSION: LazyLock = LazyLock::new(|| { + let session = vortex_array::array_session(); + vortex_fastlanes::initialize(&session); + session +}); + +/// 64 blocks of `u32`, 256 KiB in all: shorter iterations are noisy on the walltime legs. +const NUM_BLOCKS: usize = 64; + +#[vortex_bench_support::cpu_features] +#[divan::bench] +fn bitpack_blocked_compress(bencher: Bencher) { + // Block widths cycle from 1 to 16 bits, so the blocks take different packing kernels. + let bit_widths: Vec = (1..=16).cycle().take(NUM_BLOCKS).collect(); + let array = PrimitiveArray::from_iter(bit_widths.iter().flat_map(|&bit_width| { + (0..1024u32).map(move |i| i.wrapping_mul(7919) & ((1 << bit_width) - 1)) + })); + + bencher.counter(ItemsCount::new(array.len())).bench(|| { + bitpack_encode_blocked( + &array, + &bit_widths, + Some(0), + &mut SESSION.create_execution_ctx(), + ) + .unwrap() + }); +} diff --git a/encodings/fastlanes/benches/bitpack_blocked_decompress.rs b/encodings/fastlanes/benches/bitpack_blocked_decompress.rs new file mode 100644 index 00000000000..abf6d94a01d --- /dev/null +++ b/encodings/fastlanes/benches/bitpack_blocked_decompress.rs @@ -0,0 +1,69 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +//! Benchmarks decoding a `BitPacked` array whose 1024-value blocks each have their own bit width. + +#![expect(clippy::unwrap_used)] + +use std::sync::LazyLock; + +use divan::Bencher; +use divan::counter::ItemsCount; +use mimalloc::MiMalloc; +use num_traits::AsPrimitive; +use vortex_array::VortexSessionExecute; +use vortex_array::arrays::PrimitiveArray; +use vortex_array::dtype::NativePType; +use vortex_fastlanes::BitPackedArraySlotsExt; +use vortex_fastlanes::bitpack_compress::bitpack_encode_blocked; +use vortex_fastlanes::bitpack_decompress::unpack_array_blocked; +use vortex_session::VortexSession; + +#[global_allocator] +static GLOBAL: MiMalloc = MiMalloc; + +fn main() { + divan::main(); +} + +static SESSION: LazyLock = LazyLock::new(|| { + let session = vortex_array::array_session(); + vortex_fastlanes::initialize(&session); + session +}); + +/// 64 blocks per array: shorter iterations are noisy on the walltime legs. +const NUM_BLOCKS: usize = 64; + +#[vortex_bench_support::cpu_features] +#[divan::bench(types = [u32, u64])] +fn bitpack_blocked_decompress(bencher: Bencher) +where + T: NativePType, + u64: AsPrimitive, +{ + // Block widths cycle from 1 to 16 bits, so the blocks take different unpacking kernels. + let bit_widths: Vec = (1..=16).cycle().take(NUM_BLOCKS).collect(); + let values = PrimitiveArray::from_iter(bit_widths.iter().flat_map(|&bit_width| { + (0..1024u64).map(move |i| { + AsPrimitive::::as_(i.wrapping_mul(7919) & ((1 << bit_width) - 1)) + }) + })); + let array = bitpack_encode_blocked( + &values, + &bit_widths, + Some(0), + &mut SESSION.create_execution_ctx(), + ) + .unwrap(); + let offsets = array.block_offsets().unwrap().clone(); + + bencher.counter(ItemsCount::new(array.len())).bench(|| { + unpack_array_blocked( + array.as_view(), + &offsets, + &mut SESSION.create_execution_ctx(), + ) + .unwrap() + }); +} diff --git a/encodings/fastlanes/benches/bitpack_compare.rs b/encodings/fastlanes/benches/bitpack_compare.rs index ecac3446d8e..0d89e770cc1 100644 --- a/encodings/fastlanes/benches/bitpack_compare.rs +++ b/encodings/fastlanes/benches/bitpack_compare.rs @@ -30,6 +30,7 @@ use vortex_buffer::BufferMut; use vortex_fastlanes::BitPacked; use vortex_fastlanes::BitPackedArray; use vortex_fastlanes::BitPackedData; +use vortex_fastlanes::BitWidths; use vortex_session::VortexSession; #[global_allocator] @@ -58,12 +59,15 @@ const BIT_WIDTHS: &[u8] = &[4, 16]; fn page_aligned(array: BitPackedArray) -> BitPackedArray { let ptype = array.dtype().as_ptype(); let parts = BitPacked::into_parts(array); + let BitWidths::Global(bit_width) = parts.bit_widths else { + unreachable!("bitpack_encode packs every block at one bit width") + }; BitPacked::try_new( parts.packed.ensure_aligned(Alignment::new(4096)).unwrap(), ptype, parts.validity, parts.patches, - parts.bit_width, + bit_width, parts.len, parts.offset, ) diff --git a/encodings/fastlanes/benches/bitpack_compare_sweep.rs b/encodings/fastlanes/benches/bitpack_compare_sweep.rs index 1d1f91f7671..f3da868902b 100644 --- a/encodings/fastlanes/benches/bitpack_compare_sweep.rs +++ b/encodings/fastlanes/benches/bitpack_compare_sweep.rs @@ -34,6 +34,7 @@ use vortex_buffer::BufferMut; use vortex_fastlanes::BitPacked; use vortex_fastlanes::BitPackedArray; use vortex_fastlanes::BitPackedData; +use vortex_fastlanes::BitWidths; use vortex_session::VortexSession; #[global_allocator] @@ -84,12 +85,15 @@ impl_bench_int!(u8, u16, u32, u64, i8, i16, i32, i64); fn page_aligned(array: BitPackedArray) -> BitPackedArray { let ptype = array.dtype().as_ptype(); let parts = BitPacked::into_parts(array); + let BitWidths::Global(bit_width) = parts.bit_widths else { + unreachable!("bitpack_encode packs every block at one bit width") + }; BitPacked::try_new( parts.packed.ensure_aligned(Alignment::new(4096)).unwrap(), ptype, parts.validity, parts.patches, - parts.bit_width, + bit_width, parts.len, parts.offset, ) diff --git a/encodings/fastlanes/src/bitpacking/array/bitpack_compress/blocked.rs b/encodings/fastlanes/src/bitpacking/array/bitpack_compress/blocked.rs new file mode 100644 index 00000000000..62b71439553 --- /dev/null +++ b/encodings/fastlanes/src/bitpacking/array/bitpack_compress/blocked.rs @@ -0,0 +1,562 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +//! Bit-packing every 1024-value block of an array at its own bit width. + +use fastlanes::BitPacking; +use itertools::Itertools; +use num_traits::AsPrimitive; +use num_traits::PrimInt; +use vortex_array::ArrayRef; +use vortex_array::ExecutionCtx; +use vortex_array::IntoArray; +use vortex_array::arrays::PrimitiveArray; +use vortex_array::arrays::primitive::PrimitiveArrayExt; +use vortex_array::buffer::BufferHandle; +use vortex_array::dtype::IntegerPType; +use vortex_array::dtype::NativePType; +use vortex_array::dtype::PType; +use vortex_array::match_each_integer_ptype; +use vortex_array::match_each_unsigned_integer_ptype; +use vortex_array::patches::Patches; +use vortex_array::validity::Validity; +use vortex_buffer::BitBuffer; +use vortex_buffer::Buffer; +use vortex_buffer::BufferMut; +use vortex_buffer::ByteBuffer; +use vortex_error::VortexExpect; +use vortex_error::VortexResult; +use vortex_error::vortex_ensure; +use vortex_mask::AllOr; +use vortex_mask::Mask; + +use super::ensure_non_negative_integers; +use super::find_best_bit_width; +use crate::BitPacked; +use crate::BitPackedArray; +use crate::FL_CHUNK_SIZE; +use crate::bitpack_decompress; + +/// Bit-pack `array`, choosing the bit width of every 1024-value block. +/// +/// Each block uses the width that minimizes its packed size plus the cost of its exceptions. The +/// block offsets are always materialized, even when every block chooses the same width. +/// +/// # Errors +/// +/// Returns an error if `array` is not an integer array or contains negative values. +pub fn bitpack_to_best_bit_widths( + array: &PrimitiveArray, + ctx: &mut ExecutionCtx, +) -> VortexResult { + ensure_non_negative_integers(array, ctx)?; + let validity_mask = array.validity()?.execute_mask(array.len(), ctx)?; + let (bit_widths, num_exceptions) = match_each_integer_ptype!(array.ptype(), |T| { + block_bit_widths(array.as_slice::(), &validity_mask)? + }); + bitpack_encode_blocked(array, &bit_widths, Some(num_exceptions), ctx) +} + +/// Bit-pack each 1024-value block of `array` at its width in `bit_widths`. +/// +/// Valid values wider than their block's width become patches. `num_exceptions` is the number of +/// such values when the caller knows it: it sizes the patches, and zero skips gathering them. +/// +/// # Errors +/// +/// Returns an error if `array` is not an integer array or contains negative values, or if +/// `bit_widths` does not hold one width per block of at most the array's bit width. +pub fn bitpack_encode_blocked( + array: &PrimitiveArray, + bit_widths: &[u8], + num_exceptions: Option, + ctx: &mut ExecutionCtx, +) -> VortexResult { + ensure_non_negative_integers(array, ctx)?; + let num_blocks = array.len().div_ceil(FL_CHUNK_SIZE); + vortex_ensure!( + bit_widths.len() == num_blocks, + InvalidArgument: "Expected {num_blocks} bit widths, got {}", + bit_widths.len() + ); + let max_bit_width = array.ptype().bit_width(); + vortex_ensure!( + bit_widths + .iter() + .all(|&bit_width| usize::from(bit_width) <= max_bit_width), + InvalidArgument: "Bit widths must be at most {max_bit_width} for {}", + array.ptype() + ); + + // SAFETY: we check that array only contains non-negative values. + let packed = unsafe { bitpack_blocked_unchecked(array, bit_widths) }; + let patches = if num_exceptions == Some(0) { + None + } else { + let validity_mask = array.validity()?.execute_mask(array.len(), ctx)?; + gather_blocked_patches( + array, + bit_widths, + num_exceptions.unwrap_or(0), + &validity_mask, + )? + }; + + let bitpacked = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(packed), + array.ptype(), + array.validity()?, + patches, + block_offsets_from_widths(bit_widths), + array.len(), + 0, + )?; + bitpacked.statistics().inherit_from(array.statistics()); + Ok(bitpacked) +} + +/// Bitpack each 1024-value block of `array` at its width in `bit_widths`. +/// +/// # Safety +/// +/// This promotes `array` to its unsigned equivalent, like [`bitpack_unchecked`], so the caller +/// must ensure that it holds no negative values. +/// +/// [`bitpack_unchecked`]: super::bitpack_unchecked +unsafe fn bitpack_blocked_unchecked(array: &PrimitiveArray, bit_widths: &[u8]) -> ByteBuffer { + let array = array.reinterpret_cast(array.ptype().to_unsigned()); + match_each_unsigned_integer_ptype!(array.ptype(), |T| { + bitpack_blocked_primitive(array.as_slice::(), bit_widths).into_byte_buffer() + }) +} + +/// Bitpack each 1024-value block of `array` at its width in `bit_widths`, one block after another. +fn bitpack_blocked_primitive( + array: &[T], + bit_widths: &[u8], +) -> Buffer { + let block_len = |bit_width: u8| 128 * usize::from(bit_width) / size_of::(); + let mut output = + BufferMut::::with_capacity(bit_widths.iter().map(|&width| block_len(width)).sum()); + let mut pack_block = |input: &[T; FL_CHUNK_SIZE], bit_width: u8| { + let len = block_len(bit_width); + let output_len = output.len(); + // SAFETY: `input` holds 1024 values and the output window is exactly one block packed at + // its width, within the capacity reserved above. + unsafe { + output.set_len(output_len + len); + BitPacking::unchecked_pack( + usize::from(bit_width), + input, + &mut output[output_len..][..len], + ); + } + }; + + let (blocks, remainder) = array.as_chunks::(); + for (block, &bit_width) in blocks.iter().zip(bit_widths) { + pack_block(block, bit_width); + } + // Only a partial last block is zero-padded, so that the zeroing stays off the common path. + if !remainder.is_empty() { + let mut padded = [T::zero(); FL_CHUNK_SIZE]; + padded[..remainder.len()].copy_from_slice(remainder); + pack_block(&padded, bit_widths[array.len() / FL_CHUNK_SIZE]); + } + + output.freeze() +} + +/// Gather the valid values of `array` that are wider than the bit width of their 1024-value block. +fn gather_blocked_patches( + array: &PrimitiveArray, + bit_widths: &[u8], + num_exceptions_hint: usize, + validity_mask: &Mask, +) -> VortexResult> { + let patch_validity = match array.validity()? { + Validity::NonNullable => Validity::NonNullable, + _ => Validity::AllValid, + }; + let index_ptype = PType::min_unsigned_ptype_for_value(array.len() as u64); + match_each_integer_ptype!(array.ptype(), |T| { + match_each_unsigned_integer_ptype!(index_ptype, |I| { + gather_blocked_patches_typed::( + array.as_slice::(), + bit_widths, + num_exceptions_hint, + patch_validity, + validity_mask, + ) + }) + }) +} + +fn gather_blocked_patches_typed( + values: &[T], + bit_widths: &[u8], + num_exceptions_hint: usize, + patch_validity: Validity, + validity_mask: &Mask, +) -> VortexResult> +where + T: NativePType + PrimInt, + I: IntegerPType, +{ + let mut indices = BufferMut::::with_capacity(num_exceptions_hint); + let mut patch_values = BufferMut::::with_capacity(num_exceptions_hint); + let mut chunk_offsets = BufferMut::::with_capacity(bit_widths.len()); + + let mut bit_width = 0; + for ((idx, value), valid) in values.iter().enumerate().zip(validity_mask.iter()) { + if idx % FL_CHUNK_SIZE == 0 { + // Record the patch index offset and bit width of each block. + chunk_offsets.push(patch_values.len() as u64); + bit_width = bit_widths[idx / FL_CHUNK_SIZE]; + } + + if valid && (value.leading_zeros() as usize) < T::PTYPE.bit_width() - usize::from(bit_width) + { + indices.push(I::from(idx).vortex_expect("cast index from usize")); + patch_values.push(*value); + } + } + + if indices.is_empty() { + return Ok(None); + } + Ok(Some(Patches::new( + values.len(), + 0, + indices.into_array(), + PrimitiveArray::new(patch_values, patch_validity).into_array(), + Some(chunk_offsets.into_array()), + )?)) +} + +/// Byte boundaries of blocks packed at `widths`, in the narrowest unsigned type that holds them. +fn block_offsets_from_widths(widths: &[u8]) -> ArrayRef { + let end: u64 = widths.iter().map(|&width| 128 * u64::from(width)).sum(); + let ptype = PType::min_unsigned_ptype_for_value(end); + match_each_unsigned_integer_ptype!(ptype, |T| { + let mut offsets = BufferMut::::with_capacity(widths.len() + 1); + let mut offset = 0u64; + offsets.push(offset.as_()); + for &width in widths { + offset += 128 * u64::from(width); + offsets.push(offset.as_()); + } + offsets.into_array() + }) +} + +/// The width minimizing each 1024-value block's packed size plus the cost of its exceptions, and +/// the total number of exceptions those widths leave. +fn block_bit_widths( + values: &[T], + validity_mask: &Mask, +) -> VortexResult<(Vec, usize)> { + let mut histogram = vec![0usize; size_of::() * 8 + 1]; + let mut widths = Vec::with_capacity(values.len().div_ceil(FL_CHUNK_SIZE)); + let mut num_exceptions = 0; + for (block_idx, block) in values.chunks(FL_CHUNK_SIZE).enumerate() { + // The zero padding of a partial block and null values need no bits, so counting them as + // zero-width charges every block for its full packed size. + histogram.fill(0); + histogram[0] = FL_CHUNK_SIZE - block.len(); + let start = block_idx * FL_CHUNK_SIZE; + let block_validity = validity_mask.slice(start..start + block.len()); + add_bit_widths(&mut histogram, block, block_validity.bit_buffer()); + + let width = find_best_bit_width(T::PTYPE, &histogram)?; + num_exceptions += bitpack_decompress::count_exceptions(width, &histogram); + widths.push(width); + } + + Ok((widths, num_exceptions)) +} + +/// Count the bit width of each of `values` into `histogram`, counting null values as zero-width. +fn add_bit_widths( + histogram: &mut [usize], + values: &[T], + validity: AllOr<&BitBuffer>, +) { + let bit_width = |v: T| (8 * size_of::()) - (PrimInt::leading_zeros(v) as usize); + match validity { + AllOr::All => { + for v in values { + histogram[bit_width(*v)] += 1; + } + } + AllOr::None => histogram[0] += values.len(), + AllOr::Some(buffer) => { + for (is_valid, v) in buffer.iter().zip_eq(values) { + histogram[if is_valid { bit_width(*v) } else { 0 }] += 1; + } + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::LazyLock; + + use itertools::Itertools; + use rstest::rstest; + use vortex_array::VortexSessionExecute; + use vortex_array::arrays::Primitive; + use vortex_array::assert_arrays_eq; + use vortex_array::validity::Validity; + use vortex_buffer::Buffer; + use vortex_error::VortexError; + use vortex_error::vortex_bail; + use vortex_error::vortex_err; + use vortex_session::VortexSession; + + use super::*; + use crate::BitPackedData; + use crate::BitWidths; + use crate::bitpack_compress::bitpack_primitive; + use crate::bitpacking::array::BitPackedArrayExt; + + static SESSION: LazyLock = LazyLock::new(|| { + let session = vortex_array::array_session(); + crate::initialize(&session); + session + }); + + /// The materialized block offsets of a blocked array. + fn block_offsets( + array: &BitPackedArray, + ctx: &mut ExecutionCtx, + ) -> VortexResult { + let BitWidths::Blocked(offsets) = array.bit_widths() else { + vortex_bail!("expected block offsets"); + }; + offsets.execute::(ctx) + } + + /// The bit width of each block, from the distance between its boundaries. + fn block_widths(array: &BitPackedArray, ctx: &mut ExecutionCtx) -> VortexResult> { + let offsets = block_offsets(array, ctx)?; + Ok(match_each_unsigned_integer_ptype!(offsets.ptype(), |T| { + offsets + .as_slice::() + .windows(2) + .map(|pair| { + (AsPrimitive::::as_(pair[1]) - AsPrimitive::::as_(pair[0])) / 128 + }) + .collect() + })) + } + + /// Every block packed on its own at its width, one after another. + fn pack_blocks(values: &[u32], widths: &[u8]) -> Vec { + values + .chunks(1024) + .zip_eq(widths) + .flat_map(|(block, &width)| bitpack_primitive(block, width).to_vec()) + .collect() + } + + #[test] + fn best_bit_widths_choose_each_block_width() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let values: Vec = (0..1024) + .map(|i| i % 8) + .chain(std::iter::repeat_n(0, 1024)) + .chain((0..1024).map(|i| (1 << 19) + i)) + .chain((0..100).map(|i| i % 32)) + .collect(); + let array = bitpack_to_best_bit_widths( + &PrimitiveArray::from_iter(values.iter().copied()), + &mut ctx, + )?; + + assert_eq!(block_widths(&array, &mut ctx)?, [3, 0, 20, 5]); + assert_eq!(block_offsets(&array, &mut ctx)?.ptype(), PType::U16); + assert_eq!( + array.packed_slice::(), + pack_blocks(&values, &[3, 0, 20, 5]).as_slice() + ); + assert!(array.patches().is_none()); + Ok(()) + } + + #[test] + fn encode_blocked_patches_use_each_block_width() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let mut values: Vec = (0..1024) + .map(|i| i % 8) + .chain(std::iter::repeat_n(1 << 19, 1024)) + .collect(); + // 2^19 is an exception in the 3-bit block but fits the 20-bit block. + values[10] = 1 << 19; + values[20] = (1 << 20) + 1; + values[1029] = (1 << 20) + 1; + let array = bitpack_encode_blocked( + &PrimitiveArray::from_iter(values.iter().copied()), + &[3, 20], + None, + &mut ctx, + )?; + + assert_eq!(block_widths(&array, &mut ctx)?, [3, 20]); + let unsigned: Vec = values.iter().map(|v| v.cast_unsigned()).collect(); + assert_eq!( + array.packed_slice::(), + pack_blocks(&unsigned, &[3, 20]).as_slice() + ); + + let patches = array + .patches() + .ok_or_else(|| vortex_err!("expected patches"))?; + assert_arrays_eq!( + patches + .indices() + .clone() + .execute::(&mut ctx)?, + PrimitiveArray::from_iter([10u16, 20, 1029]), + &mut ctx + ); + assert_arrays_eq!( + patches + .values() + .clone() + .execute::(&mut ctx)?, + PrimitiveArray::from_iter([1i32 << 19, (1 << 20) + 1, (1 << 20) + 1]), + &mut ctx + ); + let chunk_offsets = patches + .chunk_offsets() + .as_ref() + .ok_or_else(|| vortex_err!("expected chunk offsets"))? + .clone() + .execute::(&mut ctx)?; + assert_arrays_eq!( + chunk_offsets, + PrimitiveArray::from_iter([0u64, 2]), + &mut ctx + ); + Ok(()) + } + + #[test] + fn best_bit_widths_charge_partial_blocks_for_padding() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + // Packing these values would take a full 2688-byte block; patching them takes 80 bytes. + let array = bitpack_to_best_bit_widths( + &PrimitiveArray::from_iter(vec![(1u32 << 20) + 1; 10]), + &mut ctx, + )?; + + assert_eq!(block_widths(&array, &mut ctx)?, [0]); + assert_eq!(array.packed().len(), 0); + let patches = array + .patches() + .ok_or_else(|| vortex_err!("expected patches"))?; + assert_eq!(patches.num_patches(), 10); + Ok(()) + } + + #[test] + fn best_bit_widths_materialize_uniform_offsets() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let array = bitpack_to_best_bit_widths( + &PrimitiveArray::from_iter((0..3000u32).map(|i| i % 128)), + &mut ctx, + )?; + + let BitWidths::Blocked(offsets) = array.bit_widths() else { + vortex_bail!("expected block offsets"); + }; + assert!(offsets.is::()); + assert_arrays_eq!( + block_offsets(&array, &mut ctx)?, + PrimitiveArray::from_iter([0u16, 896, 1792, 2688]), + &mut ctx + ); + Ok(()) + } + + #[test] + fn best_bit_widths_ignore_null_values() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + // Null slots hold values wider than their block, which are neither counted nor patched. + let values = PrimitiveArray::new( + (0..2048u32) + .map(|i| if i % 10 == 0 { u32::MAX } else { i % 16 }) + .collect::>(), + Validity::from_iter((0..2048).map(|i| i % 10 != 0)), + ); + let array = bitpack_to_best_bit_widths(&values, &mut ctx)?; + + assert_eq!(block_widths(&array, &mut ctx)?, [4, 4]); + assert!(array.patches().is_none()); + Ok(()) + } + + #[test] + fn best_bit_widths_of_empty_array() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let array = + bitpack_to_best_bit_widths(&PrimitiveArray::from_iter(Vec::::new()), &mut ctx)?; + + assert_arrays_eq!( + block_offsets(&array, &mut ctx)?, + PrimitiveArray::from_iter([0u8]), + &mut ctx + ); + assert_eq!(array.packed().len(), 0); + Ok(()) + } + + #[rstest] + #[case::negative(PrimitiveArray::from_iter(-5i64..5))] + #[case::float(PrimitiveArray::from_iter([1.0f32, 2.0]))] + fn encode_blocked_rejects_invalid_values(#[case] array: PrimitiveArray) { + let mut ctx = SESSION.create_execution_ctx(); + assert!(matches!( + bitpack_to_best_bit_widths(&array, &mut ctx).unwrap_err(), + VortexError::InvalidArgument(_, _) + )); + assert!(matches!( + BitPackedData::encode_blocked(&array.into_array(), &[1], &mut ctx).unwrap_err(), + VortexError::InvalidArgument(_, _) + )); + } + + #[rstest] + #[case::too_few(vec![3])] + #[case::too_many(vec![3, 3, 3])] + #[case::wider_than_type(vec![3, 33])] + fn encode_blocked_rejects_invalid_bit_widths(#[case] bit_widths: Vec) { + let array = PrimitiveArray::from_iter((0..2048u32).map(|i| i % 8)); + let err = BitPackedData::encode_blocked( + &array.into_array(), + &bit_widths, + &mut SESSION.create_execution_ctx(), + ) + .unwrap_err(); + assert!(matches!(err, VortexError::InvalidArgument(_, _))); + } + + #[test] + fn encode_blocked_accepts_native_bit_width() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let values: Vec = (0..1024).map(|i| u32::MAX - i).collect(); + let array = bitpack_encode_blocked( + &PrimitiveArray::from_iter(values.iter().copied()), + &[32], + Some(0), + &mut ctx, + )?; + assert_eq!( + array.packed_slice::(), + pack_blocks(&values, &[32]).as_slice() + ); + assert!(array.patches().is_none()); + Ok(()) + } +} diff --git a/encodings/fastlanes/src/bitpacking/array/bitpack_compress.rs b/encodings/fastlanes/src/bitpacking/array/bitpack_compress/global.rs similarity index 97% rename from encodings/fastlanes/src/bitpacking/array/bitpack_compress.rs rename to encodings/fastlanes/src/bitpacking/array/bitpack_compress/global.rs index 4fe91aad4ff..0b0371644a0 100644 --- a/encodings/fastlanes/src/bitpacking/array/bitpack_compress.rs +++ b/encodings/fastlanes/src/bitpacking/array/bitpack_compress/global.rs @@ -27,6 +27,7 @@ use vortex_error::vortex_bail; use vortex_mask::AllOr; use vortex_mask::Mask; +use super::ensure_non_negative_integers; use crate::BitPacked; use crate::BitPackedArray; use crate::bitpack_decompress; @@ -40,28 +41,18 @@ pub fn bitpack_to_best_bit_width( bitpack_encode(array, best_bit_width, Some(&bit_width_freq), ctx) } -#[expect(unused_comparisons, clippy::absurd_extreme_comparisons)] pub fn bitpack_encode( array: &PrimitiveArray, bit_width: u8, bit_width_freq: Option<&[usize]>, ctx: &mut ExecutionCtx, ) -> VortexResult { + ensure_non_negative_integers(array, ctx)?; let bit_width_freq = match bit_width_freq { Some(freq) => freq, None => &bit_width_histogram(array.as_view(), ctx)?, }; - // Check array contains no negative values. - if array.ptype().is_signed_int() { - let has_negative_values = match_each_integer_ptype!(array.ptype(), |P| { - array.statistics().compute_min::

(ctx).unwrap_or_default() < 0 - }); - if has_negative_values { - vortex_bail!(InvalidArgument: "cannot bitpack_encode array containing negative integers") - } - } - let num_exceptions = bitpack_decompress::count_exceptions(bit_width, bit_width_freq); if bit_width >= array.ptype().bit_width() as u8 { diff --git a/encodings/fastlanes/src/bitpacking/array/bitpack_compress/mod.rs b/encodings/fastlanes/src/bitpacking/array/bitpack_compress/mod.rs new file mode 100644 index 00000000000..dd5667c3c1f --- /dev/null +++ b/encodings/fastlanes/src/bitpacking/array/bitpack_compress/mod.rs @@ -0,0 +1,47 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +mod blocked; +mod global; + +pub use blocked::bitpack_encode_blocked; +pub use blocked::bitpack_to_best_bit_widths; +pub use global::bit_width_histogram; +pub use global::bitpack_encode; +pub use global::bitpack_encode_unchecked; +pub use global::bitpack_primitive; +pub use global::bitpack_to_best_bit_width; +pub use global::bitpack_unchecked; +pub use global::find_best_bit_width; +pub use global::gather_patches; +#[cfg(feature = "_test-harness")] +pub use global::test_harness; +use vortex_array::ExecutionCtx; +use vortex_array::arrays::PrimitiveArray; +use vortex_array::match_each_integer_ptype; +use vortex_error::VortexResult; +use vortex_error::vortex_bail; +use vortex_error::vortex_ensure; + +/// Return an error unless `array` holds integers that are all non-negative, which bit-packing +/// requires. +#[expect(unused_comparisons, clippy::absurd_extreme_comparisons)] +fn ensure_non_negative_integers( + array: &PrimitiveArray, + ctx: &mut ExecutionCtx, +) -> VortexResult<()> { + vortex_ensure!( + array.ptype().is_int(), + InvalidArgument: "cannot bitpack {} array", + array.ptype() + ); + if array.ptype().is_signed_int() { + let has_negative_values = match_each_integer_ptype!(array.ptype(), |P| { + array.statistics().compute_min::

(ctx).unwrap_or_default() < 0 + }); + if has_negative_values { + vortex_bail!(InvalidArgument: "cannot bitpack array containing negative integers") + } + } + Ok(()) +} diff --git a/encodings/fastlanes/src/bitpacking/array/bitpack_decompress/blocked.rs b/encodings/fastlanes/src/bitpacking/array/bitpack_decompress/blocked.rs new file mode 100644 index 00000000000..93fe7d490cf --- /dev/null +++ b/encodings/fastlanes/src/bitpacking/array/bitpack_decompress/blocked.rs @@ -0,0 +1,599 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +//! Decoding bit-packed arrays whose blocks each have their own bit width. + +use std::mem; +use std::mem::MaybeUninit; + +use fastlanes::BitPacking; +use num_traits::AsPrimitive; +use vortex_array::ArrayRef; +use vortex_array::ArrayView; +use vortex_array::ExecutionCtx; +use vortex_array::arrays::Primitive; +use vortex_array::arrays::PrimitiveArray; +use vortex_array::builders::ArrayBuilder; +use vortex_array::builders::PrimitiveBuilder; +use vortex_array::builtins::ArrayBuiltins; +use vortex_array::dtype::DType; +use vortex_array::dtype::Nullability; +use vortex_array::dtype::PType; +use vortex_array::dtype::PhysicalPType; +use vortex_array::match_each_integer_ptype; +use vortex_array::match_each_unsigned_integer_ptype; +use vortex_array::scalar::Scalar; +use vortex_buffer::Buffer; +use vortex_error::VortexResult; +use vortex_error::vortex_ensure; + +use super::global::apply_patches_to_uninit_range; +use crate::BitPacked; +use crate::BitPackedArrayExt; +use crate::FL_CHUNK_SIZE; +use crate::bitpacking::array::block_range; +use crate::bitpacking::array::validate_primitive_offsets; +use crate::unpack_iter::BitPacked as BitPackedUnpack; + +/// Unpacks a bit-packed array with block `offsets` into a primitive array. +pub fn unpack_array_blocked( + array: ArrayView<'_, BitPacked>, + offsets: &ArrayRef, + ctx: &mut ExecutionCtx, +) -> VortexResult { + match_each_integer_ptype!(array.dtype().as_ptype(), |P| { + unpack_primitive_array_blocked::

(array, offsets, ctx) + }) +} + +pub fn unpack_primitive_array_blocked( + array: ArrayView<'_, BitPacked>, + offsets: &ArrayRef, + ctx: &mut ExecutionCtx, +) -> VortexResult { + let mut builder = PrimitiveBuilder::with_capacity_in( + array.dtype().nullability(), + array.len(), + ctx.allocator(), + ); + unpack_into_primitive_builder_blocked::(array, offsets, &mut builder, ctx)?; + assert_eq!(builder.len(), array.len()); + Ok(builder.finish_into_primitive()) +} + +/// Unpack a bit-packed array with block `offsets` directly into a same-typed `PrimitiveBuilder`. +/// +/// The offsets are executed and validated before the builder is touched. Full blocks are unpacked +/// straight into the output; a partial first or last block is unpacked into a scratch block. +pub(crate) fn unpack_into_primitive_builder_blocked( + array: ArrayView<'_, BitPacked>, + offsets: &ArrayRef, + builder: &mut PrimitiveBuilder, + ctx: &mut ExecutionCtx, +) -> VortexResult<()> { + if array.is_empty() { + return Ok(()); + } + assert_eq!( + T::PTYPE, + array.dtype().as_ptype(), + "Requested type doesn't match the array ptype" + ); + + // Decoding one offset type compiles the block loop once per value type. + let offsets = offsets + .cast(DType::Primitive(PType::U64, Nullability::NonNullable))? + .execute::(ctx)?; + let num_blocks = (array.len() + usize::from(array.offset())).div_ceil(FL_CHUNK_SIZE); + vortex_ensure!( + offsets.len() == num_blocks + 1, + "Expected {} block boundaries, got {}", + num_blocks + 1, + offsets.len() + ); + let offsets = Buffer::::from_byte_buffer(offsets.buffer_handle().try_to_host_sync()?); + validate_primitive_offsets( + &offsets, + array.dtype().as_ptype().bit_width() as u64, + array.packed().len(), + )?; + + let len = array.len(); + let validity = array.validity()?.execute_mask(len, ctx)?; + let mut uninit_range = builder.uninit_range(len); + + // SAFETY: We initialize all `len` values below via `decode_blocks` and the patch loop. + unsafe { + uninit_range.append_mask(&validity); + } + + // SAFETY: `decode_blocks` writes a value to every slot in this range. + let uninit_slice = unsafe { uninit_range.slice_uninit_mut(0, len) }; + + decode_blocks(array, &offsets, uninit_slice); + + if let Some(patches) = array.patches() { + apply_patches_to_uninit_range(&mut uninit_range, &patches, ctx, |v: T| v)?; + } + + // SAFETY: A correct validity mask of `len` values was set via `append_mask`, and the same + // number of values was initialized via `decode_blocks` (and overwritten by patches). + unsafe { + uninit_range.finish(); + } + Ok(()) +} + +/// Decode the blocks between validated `offsets` into `output`. +fn decode_blocks( + array: ArrayView<'_, BitPacked>, + offsets: &[u64], + output: &mut [MaybeUninit], +) { + let packed = array.packed_slice::(); + let mut scratch = [const { MaybeUninit::::uninit() }; FL_CHUNK_SIZE]; + let base = offsets[0]; + let mut skip = usize::from(array.offset()); + let mut written = 0; + for pair in offsets.windows(2) { + // Validation bounds these differences by the packed buffer's usize length. + let start = (pair[0] - base) as usize; + let end = (pair[1] - base) as usize; + let bit_width = (end - start) / 128; + let block = &packed[start / size_of::()..end / size_of::()]; + let len = (FL_CHUNK_SIZE - skip).min(output.len() - written); + let dst = &mut output[written..][..len]; + if len == FL_CHUNK_SIZE { + // SAFETY: The boundaries have been validated against the packed length and physical + // type, and `dst` holds exactly one block. `T` and its physical type have the same + // layout. + unsafe { + let dst: &mut [T::Physical] = mem::transmute(dst); + BitPacking::unchecked_unpack(bit_width, block, dst); + } + } else { + // SAFETY: As above, with the scratch block as the destination. + unsafe { + let unpacked: &mut [T::Physical] = mem::transmute(&mut scratch[..]); + BitPacking::unchecked_unpack(bit_width, block, unpacked); + } + dst.copy_from_slice(&scratch[skip..][..len]); + } + written += len; + skip = 0; + } + debug_assert_eq!(written, output.len()); +} + +/// Decode a single value of a bit-packed array with block `offsets`, without applying patches. +/// +/// Only the boundaries of the value's block are read and validated. +pub fn unpack_single_blocked( + array: ArrayView<'_, BitPacked>, + offsets: &ArrayRef, + index: usize, + ctx: &mut ExecutionCtx, +) -> VortexResult { + let index_in_encoded = index + array.offset() as usize; + let block = index_in_encoded / FL_CHUNK_SIZE; + vortex_ensure!( + block + 1 < offsets.len(), + "BitPacked index {index} has no block boundaries" + ); + let (base, start, end) = if let Some(primitive) = offsets.as_opt::() + && primitive.buffer_handle().is_on_host() + { + match_each_unsigned_integer_ptype!(primitive.ptype(), |I| { + let offsets = primitive.as_slice::(); + ( + AsPrimitive::::as_(offsets[0]), + AsPrimitive::::as_(offsets[block]), + AsPrimitive::::as_(offsets[block + 1]), + ) + }) + } else { + let base = u64::try_from(&offsets.execute_scalar(0, ctx)?)?; + let start = if block == 0 { + base + } else { + u64::try_from(&offsets.execute_scalar(block, ctx)?)? + }; + let end = u64::try_from(&offsets.execute_scalar(block + 1, ctx)?)?; + (base, start, end) + }; + let range = block_range( + base, + start, + end, + array.dtype().as_ptype().bit_width() as u64, + array.packed().len(), + )?; + let bit_width = range.len() / 128; + match_each_integer_ptype!(array.dtype().as_ptype(), |P| { + let packed = array.packed_slice::<

::Physical>(); + let packed = &packed[range.start / size_of::

()..range.end / size_of::

()]; + // SAFETY: The boundaries validate the block's length and width against the physical type. + // The index is within this block, and signed types use the same physical bits. + let value: P = unsafe { + BitPacking::unchecked_unpack_single(bit_width, packed, index_in_encoded % FL_CHUNK_SIZE) + } + .as_(); + Ok(Scalar::primitive(value, array.dtype().nullability())) + }) +} + +#[cfg(test)] +mod tests { + use num_traits::AsPrimitive; + use rstest::rstest; + use vortex_array::IntoArray; + use vortex_array::VortexSessionExecute; + use vortex_array::arrays::ConstantArray; + use vortex_array::arrays::PrimitiveArray; + use vortex_array::assert_arrays_eq; + use vortex_array::buffer::BufferHandle; + use vortex_array::builders::ArrayBuilder; + use vortex_array::builders::PrimitiveBuilder; + use vortex_array::builtins::ArrayBuiltins; + use vortex_array::dtype::DType; + use vortex_array::dtype::NativePType; + use vortex_array::dtype::Nullability; + use vortex_array::dtype::PType; + use vortex_array::dtype::PhysicalPType; + use vortex_array::match_each_integer_ptype; + use vortex_array::match_each_unsigned_integer_ptype; + use vortex_array::patches::Patches; + use vortex_array::validity::Validity; + use vortex_buffer::BufferMut; + use vortex_buffer::ByteBuffer; + use vortex_buffer::buffer; + use vortex_error::VortexResult; + use vortex_error::vortex_bail; + + use crate::BitPacked; + use crate::BitPackedArray; + use crate::BitPackedArrayExt; + use crate::BitWidths; + use crate::FL_CHUNK_SIZE; + use crate::FoR; + use crate::bitpack_compress::bitpack_encode_blocked; + use crate::bitpack_compress::bitpack_primitive; + use crate::bitpack_compress::bitpack_to_best_bit_widths; + use crate::test::SESSION; + + fn variable( + ptype: PType, + offset: u16, + len: usize, + encoded_offsets: bool, + ) -> VortexResult<(BitPackedArray, PrimitiveArray)> { + match_each_integer_ptype!(ptype, |T| { + type U = ::Physical; + let widths = [3, 0, ptype.bit_width() as u8, 5]; + let num_blocks = (usize::from(offset) + len).div_ceil(FL_CHUNK_SIZE); + let mut packed = BufferMut::::with_capacity(num_blocks * FL_CHUNK_SIZE); + let mut values: Vec = Vec::new(); + let mut boundaries = vec![128u64]; + for &width in &widths[..num_blocks] { + let mask = 1u64 + .checked_shl(u32::from(width)) + .map_or(u64::MAX, |bit| bit - 1); + let block: Vec = (0..FL_CHUNK_SIZE) + .map(|i| AsPrimitive::::as_(u64::MAX.wrapping_sub(i as u64) & mask)) + .collect(); + packed.extend_from_slice(&bitpack_primitive(&block, width)); + values.extend(block.iter().map(|&value| AsPrimitive::::as_(value))); + boundaries.push(128 + (packed.len() * size_of::()) as u64); + } + let offsets = PrimitiveArray::from_iter(boundaries).into_array(); + let offsets = if encoded_offsets { + FoR::try_new(offsets, 0u64.into())?.into_array() + } else { + offsets + }; + let expected = + PrimitiveArray::from_iter(values[usize::from(offset)..][..len].iter().copied()); + let array = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(packed.freeze().into_byte_buffer()), + ptype, + Validity::NonNullable, + None, + offsets, + len, + offset, + )?; + Ok((array, expected)) + }) + } + + #[rstest] + #[case::full_blocks(0, 4096)] + #[case::partial_first_and_last(17, 4000)] + #[case::one_value_header(1023, 2050)] + #[case::single_full_block(0, 1024)] + #[case::single_partial_block(17, 100)] + #[case::one_value(0, 1)] + #[case::empty(0, 0)] + fn decode_variable_widths( + #[values( + PType::U8, PType::I8, PType::U16, PType::I16, PType::U32, PType::I32, PType::U64, + PType::I64 + )] + ptype: PType, + #[case] offset: u16, + #[case] len: usize, + #[values(false, true)] encoded_offsets: bool, + ) -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let (array, expected) = variable(ptype, offset, len, encoded_offsets)?; + let decoded = array + .clone() + .into_array() + .execute::(&mut ctx)?; + assert_arrays_eq!(decoded, expected, &mut ctx); + for index in [ + 0, + 1, + 1006, + 1007, + 1008, + 1023, + 1024, + 2031, + 2048, + len.saturating_sub(1), + ] { + if index < len { + assert_eq!( + array.execute_scalar(index, &mut ctx)?, + expected.execute_scalar(index, &mut ctx)? + ); + } + } + Ok(()) + } + + #[rstest] + fn decode_unsigned_offsets( + #[values(PType::U8, PType::U16, PType::U32, PType::U64)] ptype: PType, + ) -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let offsets = match_each_unsigned_integer_ptype!(ptype, |T| { + PrimitiveArray::from_iter([127u8, 255, 255].map(T::from)).into_array() + }); + let array = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(ByteBuffer::zeroed(128)), + PType::U32, + Validity::NonNullable, + None, + offsets, + 2048, + 0, + )?; + assert_arrays_eq!(array, PrimitiveArray::from_iter([0u32; 2048]), &mut ctx); + assert_eq!(array.execute_scalar(0, &mut ctx)?, 0u32.into()); + assert_eq!(array.execute_scalar(1024, &mut ctx)?, 0u32.into()); + Ok(()) + } + + #[test] + fn decode_zero_width_blocks_with_constant_offsets() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let array = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(ByteBuffer::empty()), + PType::U32, + Validity::NonNullable, + None, + ConstantArray::new(u64::MAX, 3).into_array(), + 2048, + 0, + )?; + assert_arrays_eq!(array, PrimitiveArray::from_iter([0u32; 2048]), &mut ctx); + Ok(()) + } + + #[test] + fn decode_with_nulls_and_patches() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let (array, expected) = variable(PType::U32, 17, 3100, true)?; + let mut values: Vec<_> = expected + .as_slice::() + .iter() + .copied() + .map(Some) + .collect(); + let indices = [0u32, 1007, 1008, 2031, 2032, 3099]; + for &index in &indices { + values[index as usize] = Some(999); + } + values[1008] = None; + values[2048] = None; + let expected = PrimitiveArray::from_option_iter(values.iter().copied()); + let patches = Patches::new( + array.len(), + 113, + PrimitiveArray::from_iter(indices.map(|i| i + 113)).into_array(), + PrimitiveArray::new(buffer![999u32; 6], Validity::AllValid).into_array(), + None, + )?; + let BitWidths::Blocked(block_offsets) = array.bit_widths() else { + vortex_bail!("expected block offsets"); + }; + let array = BitPacked::try_new_with_block_offsets( + array.packed().clone(), + PType::U32, + expected.validity()?, + Some(patches), + block_offsets, + array.len(), + array.offset(), + )?; + let mut builder = PrimitiveBuilder::::with_capacity_in( + Nullability::Nullable, + array.len() + 2, + ctx.allocator(), + ); + builder.append_null(); + array.append_to_builder(&mut builder, &mut ctx)?; + builder.append_value(7); + let appended = + PrimitiveArray::from_option_iter([None].into_iter().chain(values).chain([Some(7)])); + assert_arrays_eq!(builder.finish_into_primitive(), appended, &mut ctx); + for index in [0, 1007, 1008, 2031, 2032, 2048, 3099] { + assert_eq!( + array.execute_scalar(index, &mut ctx)?, + expected.execute_scalar(index, &mut ctx)? + ); + } + Ok(()) + } + + #[test] + fn compute_falls_back_to_variable_width_decode() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let (array, expected) = variable(PType::U32, 17, 4000, false)?; + let array = array.into_array(); + let expected = expected.into_array(); + assert_arrays_eq!( + array.slice(1000..2200)?, + expected.slice(1000..2200)?, + &mut ctx + ); + let indices = buffer![3999u32, 0, 1007, 2048, 1024, 2048].into_array(); + assert_arrays_eq!( + array.take(indices.clone())?, + expected.take(indices)?, + &mut ctx + ); + let dtype = DType::Primitive(PType::U64, Nullability::NonNullable); + assert_arrays_eq!(array.cast(dtype.clone())?, expected.cast(dtype)?, &mut ctx); + Ok(()) + } + + #[rstest] + #[case::before_base([128, 0, 256])] + #[case::decreasing([0, 256, 128])] + #[case::misaligned_start([0, 1, 129])] + #[case::misaligned_end([0, 128, 255])] + #[case::past_end([0, 128, 4224])] + #[case::too_wide([0, 0, 1152])] + fn scalar_rejects_invalid_block(#[case] offsets: [u64; 3]) -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let array = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(ByteBuffer::zeroed(4096)), + PType::U8, + Validity::NonNullable, + None, + FoR::try_new(PrimitiveArray::from_iter(offsets).into_array(), 0u64.into())? + .into_array(), + 2048, + 0, + )?; + assert!(array.execute_scalar(1024, &mut ctx).is_err()); + Ok(()) + } + + #[test] + fn scalar_validates_only_the_selected_block() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let offsets = FoR::try_new(buffer![128u64, 256, 255].into_array(), 0u64.into())?; + let array = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(ByteBuffer::zeroed(128)), + PType::U32, + Validity::NonNullable, + None, + offsets.into_array(), + 2048, + 0, + )?; + assert_eq!(array.execute_scalar(0, &mut ctx)?, 0u32.into()); + assert!(array.execute_scalar(1024, &mut ctx).is_err()); + Ok(()) + } + + /// Non-negative values whose 1024-value blocks need different bit widths, with outliers in the + /// first block. + fn blocked_values(len: usize) -> Vec + where + u64: AsPrimitive, + { + let max_bits = + u32::try_from(T::PTYPE.bit_width()).unwrap() - u32::from(T::PTYPE.is_signed_int()); + let widths = [3, 0, max_bits.min(12), 5]; + (0..len) + .map(|i| { + let width = if i < FL_CHUNK_SIZE && i % 97 == 0 { + max_bits + } else { + widths[(i / FL_CHUNK_SIZE) % widths.len()] + }; + let mask = 1u64.checked_shl(width).map_or(u64::MAX, |bit| bit - 1); + (u64::MAX.wrapping_sub(i as u64) & mask).as_() + }) + .collect() + } + + #[rstest] + fn round_trip_best_bit_widths( + #[values( + PType::U8, PType::I8, PType::U16, PType::I16, PType::U32, PType::I32, PType::U64, + PType::I64 + )] + ptype: PType, + #[values(1, 1024, 4000)] len: usize, + #[values(false, true)] nullable: bool, + ) -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let array = match_each_integer_ptype!(ptype, |T| { + let values = blocked_values::(len); + if nullable { + PrimitiveArray::from_option_iter( + values + .into_iter() + .enumerate() + .map(|(i, value)| (i % 7 != 3).then_some(value)), + ) + } else { + PrimitiveArray::from_iter(values) + } + }); + let encoded = bitpack_to_best_bit_widths(&array, &mut ctx)?; + assert!(matches!(encoded.bit_widths(), BitWidths::Blocked(_))); + assert_arrays_eq!(encoded, array, &mut ctx); + for index in [0, len / 2, len - 1] { + assert_eq!( + encoded.execute_scalar(index, &mut ctx)?, + array.execute_scalar(index, &mut ctx)? + ); + } + Ok(()) + } + + #[rstest] + fn round_trip_explicit_bit_widths( + #[values(PType::U8, PType::I16, PType::U32, PType::I64)] ptype: PType, + ) -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let array = match_each_integer_ptype!(ptype, |T| { + PrimitiveArray::from_iter(blocked_values::(4000)) + }); + // Zero and native widths, plus narrow widths that turn most values into patches. + let bit_widths = [0, u8::try_from(ptype.bit_width())?, 1, 3]; + let encoded = bitpack_encode_blocked(&array, &bit_widths, None, &mut ctx)?; + assert_arrays_eq!(encoded, array, &mut ctx); + Ok(()) + } + + #[test] + fn for_decodes_blocked_child() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let values = blocked_values::(4000); + let encoded = + bitpack_to_best_bit_widths(&PrimitiveArray::from_iter(values.clone()), &mut ctx)?; + let array = FoR::try_new(encoded.into_array(), 1000u32.into())?; + let expected = + PrimitiveArray::from_iter(values.into_iter().map(|value| value.wrapping_add(1000))); + assert_arrays_eq!(array, expected, &mut ctx); + Ok(()) + } +} diff --git a/encodings/fastlanes/src/bitpacking/array/bitpack_decompress.rs b/encodings/fastlanes/src/bitpacking/array/bitpack_decompress/global.rs similarity index 98% rename from encodings/fastlanes/src/bitpacking/array/bitpack_decompress.rs rename to encodings/fastlanes/src/bitpacking/array/bitpack_decompress/global.rs index 0684aea5e6a..a8d1bb6834c 100644 --- a/encodings/fastlanes/src/bitpacking/array/bitpack_decompress.rs +++ b/encodings/fastlanes/src/bitpacking/array/bitpack_decompress/global.rs @@ -17,11 +17,12 @@ use vortex_array::match_each_integer_ptype; use vortex_array::match_each_unsigned_integer_ptype; use vortex_array::patches::Patches; use vortex_array::scalar::Scalar; -use vortex_error::VortexExpect; use vortex_error::VortexResult; +use vortex_error::vortex_bail; use crate::BitPacked; use crate::BitPackedArrayExt; +use crate::BitWidths; use crate::FL_CHUNK_SIZE; use crate::unpack_iter::BitPacked as BitPackedUnpack; use crate::unpack_iter::BitUnpackedChunks; @@ -114,18 +115,19 @@ where } let len = array.len(); + let mut scratch = [const { MaybeUninit::::uninit() }; FL_CHUNK_SIZE]; + let mut chunks = array.unpacked_chunks::(&mut scratch)?; + let validity = array.validity()?.execute_mask(len, ctx)?; let mut uninit_range = builder.uninit_range(len); // SAFETY: We initialize all `len` values below via `decode` and the patch loop. unsafe { - uninit_range.append_mask(&array.validity()?.execute_mask(len, ctx)?); + uninit_range.append_mask(&validity); } // SAFETY: `decode` writes a value to every slot in this range. let uninit_slice = unsafe { uninit_range.slice_uninit_mut(0, len) }; - let mut scratch = [const { MaybeUninit::::uninit() }; FL_CHUNK_SIZE]; - let mut chunks = array.unpacked_chunks::(&mut scratch)?; decode(&mut chunks, uninit_slice, &map); if let Some(patches) = array.patches() { @@ -164,8 +166,11 @@ pub(crate) fn apply_patches_to_uninit_range, index: usize) -> Scalar { - let bit_width = array.bit_width() as usize; +pub fn unpack_single(array: ArrayView<'_, BitPacked>, index: usize) -> VortexResult { + let BitWidths::Global(bit_width) = array.bit_widths() else { + vortex_bail!("BitPacked array has per-block bit widths"); + }; + let bit_width = bit_width as usize; let ptype = array.dtype().as_ptype(); // let packed = array.packed().into_primitive()?; let index_in_encoded = index + array.offset() as usize; @@ -176,7 +181,7 @@ pub fn unpack_single(array: ArrayView<'_, BitPacked>, index: usize) -> Scalar { } }); // Cast to fix signedness and nullability - scalar.cast(array.dtype()).vortex_expect("cast failure") + scalar.cast(array.dtype()) } /// # Safety @@ -227,6 +232,7 @@ mod tests { use vortex_buffer::Buffer; use vortex_buffer::BufferMut; use vortex_buffer::buffer; + use vortex_error::VortexExpect; use vortex_session::VortexSession; use super::*; @@ -259,7 +265,7 @@ mod tests { .iter() .enumerate() .for_each(|(i, v)| { - let scalar: u16 = (&unpack_single(compressed.as_view(), i)) + let scalar: u16 = (&unpack_single(compressed.as_view(), i).unwrap()) .try_into() .unwrap(); assert_eq!(scalar, *v); diff --git a/encodings/fastlanes/src/bitpacking/array/bitpack_decompress/mod.rs b/encodings/fastlanes/src/bitpacking/array/bitpack_decompress/mod.rs new file mode 100644 index 00000000000..bbc4ca4e713 --- /dev/null +++ b/encodings/fastlanes/src/bitpacking/array/bitpack_decompress/mod.rs @@ -0,0 +1,17 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +mod blocked; +mod global; + +pub use blocked::unpack_array_blocked; +pub(crate) use blocked::unpack_into_primitive_builder_blocked; +pub use blocked::unpack_primitive_array_blocked; +pub use blocked::unpack_single_blocked; +pub use global::count_exceptions; +pub use global::unpack_array; +pub(crate) use global::unpack_into_primitive_builder; +pub(crate) use global::unpack_map_into_builder; +pub use global::unpack_primitive_array; +pub use global::unpack_single; +pub use global::unpack_single_primitive; diff --git a/encodings/fastlanes/src/bitpacking/array/mod.rs b/encodings/fastlanes/src/bitpacking/array/mod.rs index 05cbee8b3ef..b70269d5515 100644 --- a/encodings/fastlanes/src/bitpacking/array/mod.rs +++ b/encodings/fastlanes/src/bitpacking/array/mod.rs @@ -4,6 +4,7 @@ use std::fmt::Display; use std::fmt::Formatter; use std::mem::MaybeUninit; +use std::ops::Range; use fastlanes::BitPacking; use vortex_array::ArrayRef; @@ -16,22 +17,29 @@ use vortex_array::buffer::BufferHandle; use vortex_array::dtype::DType; use vortex_array::dtype::NativePType; use vortex_array::dtype::PType; +use vortex_array::match_each_unsigned_integer_ptype; use vortex_array::patches::PatchSlotIndices; use vortex_array::patches::Patches; use vortex_array::patches::PatchesData; use vortex_array::validity::Validity; use vortex_array::vtable::child_to_validity; +use vortex_error::VortexExpect; use vortex_error::VortexResult; use vortex_error::vortex_ensure; use vortex_error::vortex_err; +use vortex_error::vortex_panic; pub mod bitpack_compress; pub mod bitpack_decompress; pub mod unpack_iter; +#[cfg(test)] +mod tests; + use crate::BitPackedArray; use crate::FL_CHUNK_SIZE; use crate::bitpack_compress::bitpack_encode; +use crate::bitpack_compress::bitpack_encode_blocked; use crate::unpack_iter::BitPacked as BitPackedIter; use crate::unpack_iter::BitUnpackedChunks; @@ -49,6 +57,11 @@ pub struct BitPackedSlots { /// The validity bitmap indicating which elements are non-null. #[slot(3)] pub validity_child: Option, + /// Byte boundaries of the packed blocks as non-nullable unsigned integers, including one + /// trailing boundary, when the blocks have no global bit width. + /// Block `i` is packed at `(block_offsets[i + 1] - block_offsets[i]) / 128` bits. + #[slot(4)] + pub block_offsets: Option, } pub(crate) const PATCH_SLOTS: PatchSlotIndices = PatchSlotIndices { @@ -57,9 +70,123 @@ pub(crate) const PATCH_SLOTS: PatchSlotIndices = PatchSlotIndices { chunk_offsets: BitPackedSlots::PATCH_CHUNK_OFFSETS, }; +/// Check that `offsets` holds `num_blocks + 1` non-nullable unsigned boundaries. +/// +/// Debug builds also assert that host-resident boundaries span `packed_len` bytes, each block a +/// whole number of 128-byte rows with a bit width supported by `ptype`. Release builds don't check +/// the boundary values, so decoders must bounds-check them. +pub(crate) fn validate_block_offsets( + offsets: &ArrayRef, + ptype: PType, + num_blocks: usize, + packed_len: usize, +) -> VortexResult<()> { + vortex_ensure!( + offsets.dtype().is_unsigned_int() && !offsets.dtype().is_nullable(), + "Expected non-nullable unsigned integer block offsets, got {}", + offsets.dtype() + ); + vortex_ensure!( + offsets.len() == num_blocks + 1, + "Expected {} block boundaries, got {}", + num_blocks + 1, + offsets.len() + ); + if cfg!(debug_assertions) + && let Some(primitive) = offsets.as_opt::() + && primitive.buffer_handle().is_on_host() + { + let max_bit_width = ptype.bit_width() as u64; + match_each_unsigned_integer_ptype!(primitive.ptype(), |T| { + validate_primitive_offsets(primitive.as_slice::(), max_bit_width, packed_len) + .vortex_expect("invalid BitPacked block offsets") + }) + } + Ok(()) +} + +/// Check that each block between `boundaries` is a whole number of 128-byte rows of at most +/// `max_bit_width` bits, and that the boundaries span `packed_len` bytes. +fn validate_primitive_offsets( + boundaries: &[T], + max_bit_width: u64, + packed_len: usize, +) -> VortexResult<()> +where + u64: From, +{ + let base = boundaries.first().map_or(0, |&first| u64::from(first)); + for pair in boundaries.windows(2) { + block_range( + base, + u64::from(pair[0]), + u64::from(pair[1]), + max_bit_width, + packed_len, + )?; + } + let span = match boundaries { + [first, .., last] => u64::from(*last) - u64::from(*first), + _ => 0, + }; + vortex_ensure!( + span == packed_len as u64, + "Block offsets span {span} bytes, but the packed buffer has {packed_len}" + ); + Ok(()) +} + +/// Return the bytes of the block between boundaries `start` and `end`, relative to the first +/// boundary `base`. +/// +/// The block must lie within `packed_len` bytes and be a whole number of 128-byte rows of at most +/// `max_bit_width` bits. +fn block_range( + base: u64, + start: u64, + end: u64, + max_bit_width: u64, + packed_len: usize, +) -> VortexResult> { + vortex_ensure!( + base <= start && start <= end, + "Block boundaries {start} and {end} are decreasing (base {base})" + ); + vortex_ensure!( + end - base <= packed_len as u64, + "Block boundaries {start} and {end} are outside the packed buffer (base {base})" + ); + let size = end - start; + vortex_ensure!( + (start - base).is_multiple_of(128) + && size.is_multiple_of(128) + && size / 128 <= max_bit_width, + "Block boundaries {start} and {end} do not hold a supported bit width (at most {max_bit_width} bits)" + ); + // Both differences are at most `packed_len`, so they fit in `usize`. + Ok((start - base) as usize..(end - base) as usize) +} + +/// How the blocks of a [`BitPackedArray`] are packed. +#[derive(Clone, Debug)] +pub enum BitWidths { + /// Every block is packed at this bit width. + Global(u8), + /// Byte boundaries of the packed blocks, from which each block's bit width is derived. + Blocked(ArrayRef), +} + +impl BitWidths { + /// Returns `true` if every block is packed at one bit width. + #[inline] + pub fn is_global(&self) -> bool { + matches!(self, Self::Global(_)) + } +} + pub struct BitPackedDataParts { pub offset: u16, - pub bit_width: u8, + pub bit_widths: BitWidths, pub len: usize, pub packed: BufferHandle, pub patches: Option, @@ -71,7 +198,9 @@ pub struct BitPackedData { /// The offset within the first block (created with a slice). /// 0 <= offset < 1024 pub(super) offset: u16, - pub(super) bit_width: u8, + /// The bit width shared by every block, or `None` when the block offsets child holds each + /// block's boundaries. + pub(super) global_bit_width: Option, pub(super) packed: BufferHandle, /// Patch metadata for reconstructing Patches from slots. pub(super) patches_data: Option, @@ -79,7 +208,10 @@ pub struct BitPackedData { impl Display for BitPackedData { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - write!(f, "bit_width: {}, offset: {}", self.bit_width, self.offset) + match self.global_bit_width { + Some(bit_width) => write!(f, "bit_width: {}, offset: {}", bit_width, self.offset), + None => write!(f, "offset: {}", self.offset), + } } } @@ -139,7 +271,26 @@ impl BitPackedData { Ok(Self { offset, - bit_width, + global_bit_width: Some(bit_width), + packed, + patches_data: patches.as_ref().map(PatchesData::from_patches), + }) + } + + /// Create the packed payload for blocks whose bit widths come from a block offsets child. + pub(crate) fn try_new_blocked( + packed: BufferHandle, + patches: Option, + offset: u16, + ) -> VortexResult { + vortex_ensure!( + offset < 1024, + "Offset must be less than the full block i.e., 1024, got {offset}" + ); + + Ok(Self { + offset, + global_bit_width: None, packed, patches_data: patches.as_ref().map(PatchesData::from_patches), }) @@ -150,12 +301,11 @@ impl BitPackedData { ptype: PType, validity: &Validity, patches: Option<&Patches>, - bit_width: u8, + bit_width: Option, length: usize, offset: u16, ) -> VortexResult<()> { vortex_ensure!(ptype.is_int(), MismatchedTypes: "integer", ptype); - vortex_ensure!(bit_width <= 64, "Unsupported bit width {bit_width}"); if let Some(validity_len) = validity.maybe_len() { vortex_ensure!( @@ -169,15 +319,21 @@ impl BitPackedData { Self::validate_patches(patches, ptype, length)?; } - // Validate packed buffer - let expected_packed_len = - (length + offset as usize).div_ceil(1024) * (128 * bit_width as usize); - vortex_ensure!( - packed.len() == expected_packed_len, - "Expected {} packed bytes, got {}", - expected_packed_len, - packed.len() - ); + // Validate packed buffer. Block offsets are only checked against it in debug builds. + if let Some(bit_width) = bit_width { + vortex_ensure!( + usize::from(bit_width) <= ptype.bit_width(), + "Unsupported bit width {bit_width} for {ptype}" + ); + let expected_packed_len = + (length + offset as usize).div_ceil(1024) * (128 * bit_width as usize); + vortex_ensure!( + packed.len() == expected_packed_len, + "Expected {} packed bytes, got {}", + expected_packed_len, + packed.len() + ); + } Ok(()) } @@ -239,12 +395,6 @@ impl BitPackedData { BitUnpackedChunks::try_new(self, len, scratch) } - /// Bit-width of the packed values - #[inline] - pub fn bit_width(&self) -> u8 { - self.bit_width - } - #[inline] pub fn offset(&self) -> u16 { self.offset @@ -273,12 +423,26 @@ impl BitPackedData { bitpack_encode(&parray, bit_width, None, ctx) } - /// Calculate the maximum value that **can** be contained by this array, given its bit-width. + /// Bit-pack an array of primitive integers, packing each 1024-value block at its width in + /// `bit_widths`. /// - /// Note that this value need not actually be present in the array. - #[inline] - pub fn max_packed_value(&self) -> usize { - (1 << self.bit_width()) - 1 + /// Values wider than their block's width become patches. + /// + /// # Errors + /// + /// If the provided array is not a primitive integer array or contains negative values, or if + /// `bit_widths` does not hold one width per block of at most the array's bit width, an error + /// will be returned. + pub fn encode_blocked( + array: &ArrayRef, + bit_widths: &[u8], + ctx: &mut ExecutionCtx, + ) -> VortexResult { + let parray: PrimitiveArray = array + .clone() + .try_downcast::() + .map_err(|a| vortex_err!(InvalidArgument: "Bitpacking can only encode primitive arrays, got {}", a.encoding_id()))?; + bitpack_encode_blocked(&parray, bit_widths, None, ctx) } } @@ -288,9 +452,17 @@ pub trait BitPackedArrayExt: BitPackedArraySlotsExt { BitPackedData::packed(self) } + /// How the blocks are packed: at one global bit width, or at the widths implied by the block + /// offsets. #[inline] - fn bit_width(&self) -> u8 { - BitPackedData::bit_width(self) + fn bit_widths(&self) -> BitWidths { + match (self.global_bit_width, self.block_offsets()) { + (Some(bit_width), None) => BitWidths::Global(bit_width), + (None, Some(block_offsets)) => BitWidths::Blocked(block_offsets.clone()), + _ => vortex_panic!( + "BitPacked must have exactly one of a global bit width and block offsets" + ), + } } #[inline] diff --git a/encodings/fastlanes/src/bitpacking/array/tests.rs b/encodings/fastlanes/src/bitpacking/array/tests.rs new file mode 100644 index 00000000000..3489d0b5f68 --- /dev/null +++ b/encodings/fastlanes/src/bitpacking/array/tests.rs @@ -0,0 +1,390 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +//! Tests for construction and validation of the global bit width and block offsets. + +use std::sync::LazyLock; + +use rstest::rstest; +use vortex_array::Array; +use vortex_array::ArrayParts; +use vortex_array::ArrayRef; +use vortex_array::ArraySlots; +use vortex_array::IntoArray; +use vortex_array::VortexSessionExecute; +use vortex_array::arrays::PrimitiveArray; +use vortex_array::assert_arrays_eq; +use vortex_array::buffer::BufferHandle; +use vortex_array::builders::ArrayBuilder; +use vortex_array::builders::PrimitiveBuilder; +use vortex_array::dtype::DType; +use vortex_array::dtype::Nullability; +use vortex_array::dtype::PType; +use vortex_array::match_each_unsigned_integer_ptype; +use vortex_array::scalar_fn::fns::cast::CastKernel; +use vortex_array::scalar_fn::fns::cast::CastReduce; +use vortex_array::validity::Validity; +use vortex_buffer::ByteBuffer; +use vortex_buffer::buffer; +use vortex_error::VortexResult; +use vortex_error::vortex_bail; +use vortex_session::VortexSession; + +use crate::BitPacked; +use crate::BitPackedArray; +use crate::BitPackedArrayExt; +use crate::BitPackedArraySlotsExt; +use crate::BitPackedData; +use crate::BitWidths; +use crate::FoR; +use crate::bitpacking::bitpack_compress::bitpack_to_best_bit_width; + +static SESSION: LazyLock = LazyLock::new(|| { + let session = vortex_array::array_session(); + crate::initialize(&session); + session +}); + +fn encode(values: &[u32]) -> VortexResult { + let mut ctx = SESSION.create_execution_ctx(); + bitpack_to_best_bit_width(&PrimitiveArray::from_iter(values.iter().copied()), &mut ctx) +} + +fn uniform() -> VortexResult { + encode(&(0..3000u32).map(|i| i % 128).collect::>()) +} + +fn with_block_offsets(array: &BitPackedArray, offsets: ArrayRef) -> VortexResult { + BitPacked::try_new_with_block_offsets( + array.packed().clone(), + array.dtype().as_ptype(), + array.validity()?, + array.patches(), + offsets, + array.len(), + array.offset(), + ) +} + +#[test] +fn global_bit_width_has_no_block_offsets() -> VortexResult<()> { + let uniform = uniform()?; + assert!(matches!(uniform.bit_widths(), BitWidths::Global(7))); + assert!(uniform.block_offsets().is_none()); + Ok(()) +} + +#[rstest] +#[case::both(true, true)] +#[case::neither(false, false)] +fn global_bit_width_and_block_offsets_are_exclusive( + #[case] constant: bool, + #[case] offsets: bool, +) -> VortexResult<()> { + let packed = BufferHandle::new_host(ByteBuffer::zeroed(128)); + let data = if constant { + BitPackedData::try_new(packed, None, 1, 0)? + } else { + BitPackedData::try_new_blocked(packed, None, 0)? + }; + let block_offsets = offsets.then(|| buffer![0u64, 128].into_array()); + let slots: ArraySlots = [None, None, None, None, block_offsets] + .into_iter() + .collect(); + let dtype = DType::Primitive(PType::U32, Nullability::NonNullable); + assert!( + Array::::try_from_parts( + ArrayParts::new(BitPacked, dtype, 1024, data).with_slots(slots) + ) + .is_err() + ); + Ok(()) +} + +#[rstest] +fn unsigned_block_offsets_are_supported( + #[values(PType::U8, PType::U16, PType::U32, PType::U64)] ptype: PType, +) -> VortexResult<()> { + let offsets = match_each_unsigned_integer_ptype!(ptype, |T| { + PrimitiveArray::from_iter([127 as T, 255 as T]).into_array() + }); + let array = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(ByteBuffer::zeroed(128)), + PType::U32, + Validity::NonNullable, + None, + offsets.clone(), + 1024, + 0, + )?; + assert!( + array + .block_offsets() + .is_some_and(|block_offsets| ArrayRef::ptr_eq(&offsets, block_offsets)) + ); + assert_arrays_eq!( + array, + PrimitiveArray::from_iter([0u32; 1024]), + &mut SESSION.create_execution_ctx() + ); + Ok(()) +} + +#[cfg(debug_assertions)] +#[rstest] +#[case::unaligned([0, 127])] +#[case::decreasing([128, 0])] +#[case::wrong_span([0, 0])] +#[should_panic(expected = "invalid BitPacked block offsets")] +fn invalid_unsigned_block_boundaries_panic_in_debug( + #[case] boundaries: [u8; 2], + #[values(PType::U8, PType::U16, PType::U32, PType::U64)] ptype: PType, +) { + let offsets = match_each_unsigned_integer_ptype!(ptype, |T| { + PrimitiveArray::from_iter(boundaries.map(T::from)).into_array() + }); + BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(ByteBuffer::zeroed(128)), + PType::U32, + Validity::NonNullable, + None, + offsets, + 1024, + 0, + ) + .unwrap(); +} + +#[rstest] +#[case::equal_steps(buffer![0u64, 896, 1792, 2688])] +#[case::different_widths(buffer![0u64, 384, 1408, 2688])] +fn block_offsets_have_no_constant_width( + #[case] offsets: vortex_buffer::Buffer, +) -> VortexResult<()> { + let array = with_block_offsets(&uniform()?, offsets.into_array())?; + assert!(matches!(array.bit_widths(), BitWidths::Blocked(_))); + Ok(()) +} + +#[test] +fn equal_block_offsets_decode() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let original = uniform()?; + let blocked = with_block_offsets(&original, buffer![128u64, 1024, 1920, 2816].into_array())?; + assert_arrays_eq!(original, blocked, &mut ctx); + Ok(()) +} + +#[rstest] +#[case::equal_steps(buffer![0u64, 896, 1792, 2688])] +#[case::different_widths(buffer![0u64, 384, 1408, 2688])] +fn casts_with_block_offsets_decline( + #[case] offsets: vortex_buffer::Buffer, + #[values( + DType::Primitive(PType::U32, Nullability::Nullable), + DType::Primitive(PType::U64, Nullability::NonNullable) + )] + dtype: DType, +) -> VortexResult<()> { + let array = with_block_offsets(&uniform()?, offsets.into_array())?; + let mut ctx = SESSION.create_execution_ctx(); + + assert!(::cast(array.as_view(), &dtype)?.is_none()); + assert!(::cast(array.as_view(), &dtype, &mut ctx)?.is_none()); + Ok(()) +} + +#[rstest] +#[case::too_few_boundaries(buffer![0u64, 896, 1792].into_array())] +#[case::signed(buffer![0i32, 896, 1792, 2688].into_array())] +#[case::float(buffer![0f32, 896.0, 1792.0, 2688.0].into_array())] +#[case::nullable( + PrimitiveArray::from_option_iter([Some(0u32), Some(896), Some(1792), Some(2688)]).into_array() +)] +fn invalid_block_offsets_are_rejected(#[case] offsets: ArrayRef) -> VortexResult<()> { + assert!(with_block_offsets(&uniform()?, offsets).is_err()); + Ok(()) +} + +#[cfg(debug_assertions)] +#[rstest] +#[case::unaligned_block(buffer![0u64, 896, 1791, 2688])] +#[case::decreasing(buffer![0u64, 896, 768, 2688])] +#[case::span_disagrees_with_packed_len(buffer![0u64, 768, 1536, 2304])] +#[should_panic(expected = "invalid BitPacked block offsets")] +fn invalid_block_boundaries_panic_in_debug(#[case] offsets: vortex_buffer::Buffer) { + with_block_offsets(&uniform().unwrap(), offsets.into_array()).unwrap(); +} + +/// One block of `ptype` values packed at `bit_width`, either globally or through block offsets. +fn single_block(ptype: PType, bit_width: u8, block_offsets: bool) -> VortexResult { + let packed = BufferHandle::new_host(ByteBuffer::zeroed(128 * usize::from(bit_width))); + if block_offsets { + let end = 128 * u64::from(bit_width); + BitPacked::try_new_with_block_offsets( + packed, + ptype, + Validity::NonNullable, + None, + buffer![0u64, end].into_array(), + 1024, + 0, + ) + } else { + BitPacked::try_new( + packed, + ptype, + Validity::NonNullable, + None, + bit_width, + 1024, + 0, + ) + } +} + +#[rstest] +fn bit_width_must_fit_ptype( + #[values( + PType::U8, PType::I8, PType::U16, PType::I16, PType::U32, PType::I32, PType::U64, + PType::I64 + )] + ptype: PType, + #[values(false, true)] too_wide: bool, +) -> VortexResult<()> { + let bit_width = u8::try_from(ptype.bit_width())? + u8::from(too_wide); + assert_eq!(single_block(ptype, bit_width, false).is_err(), too_wide); + Ok(()) +} + +#[rstest] +fn block_offsets_can_fill_ptype( + #[values( + PType::U8, PType::I8, PType::U16, PType::I16, PType::U32, PType::I32, PType::U64, + PType::I64 + )] + ptype: PType, +) -> VortexResult<()> { + single_block(ptype, u8::try_from(ptype.bit_width())?, true)?; + Ok(()) +} + +#[cfg(debug_assertions)] +#[rstest] +#[should_panic(expected = "invalid BitPacked block offsets")] +fn too_wide_block_panics_in_debug( + #[values( + PType::U8, PType::I8, PType::U16, PType::I16, PType::U32, PType::I32, PType::U64, + PType::I64 + )] + ptype: PType, +) { + let bit_width = u8::try_from(ptype.bit_width()).unwrap() + 1; + single_block(ptype, bit_width, true).unwrap(); +} + +#[rstest] +#[case::unaligned(buffer![0u64, 384, 1023], 1024, "supported bit width")] +#[case::decreasing(buffer![0u64, 640, 128], 1024, "decreasing")] +#[case::wrong_span(buffer![0u64, 512, 896], 1024, "span")] +#[case::too_wide(buffer![0u64, 4224, 4352], 4352, "supported bit width")] +fn invalid_encoded_offsets_leave_builder_unchanged( + #[case] offsets: vortex_buffer::Buffer, + #[case] packed_len: usize, + #[case] error: &str, +) -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let array = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(ByteBuffer::zeroed(packed_len)), + PType::U32, + Validity::AllValid, + None, + FoR::try_new(offsets.into_array(), 0u64.into())?.into_array(), + 2048, + 0, + )?; + let mut builder = PrimitiveBuilder::::with_capacity_in( + Nullability::Nullable, + array.len() + 2, + ctx.allocator(), + ); + builder.append_null(); + let err = array + .append_to_builder(&mut builder, &mut ctx) + .unwrap_err() + .to_string(); + assert!(err.contains(error), "{err}"); + builder.append_value(7); + assert_arrays_eq!( + builder.finish_into_primitive(), + PrimitiveArray::from_option_iter([None, Some(7u32)]), + &mut ctx + ); + Ok(()) +} + +#[test] +fn construct_blocks_without_a_uniform_width() -> VortexResult<()> { + // Two blocks at widths 3 and 4 occupy 896 bytes, which no constant width can represent. + let array = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(ByteBuffer::zeroed(896)), + PType::U32, + Validity::NonNullable, + None, + buffer![0u64, 384, 896].into_array(), + 2048, + 0, + )?; + assert!(matches!(array.bit_widths(), BitWidths::Blocked(_))); + Ok(()) +} + +#[test] +fn empty_and_zero_width_arrays() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + assert!(matches!(encode(&[])?.bit_widths(), BitWidths::Global(0))); + let zeros = encode(&vec![0u32; 2049])?; + assert!(matches!(zeros.bit_widths(), BitWidths::Global(0))); + assert_eq!(zeros.packed().len(), 0); + assert_arrays_eq!(zeros, PrimitiveArray::from_iter(vec![0u32; 2049]), &mut ctx); + Ok(()) +} + +#[test] +fn into_parts_preserves_global_bit_width() -> VortexResult<()> { + let parts = BitPacked::into_parts(uniform()?); + assert!(matches!(parts.bit_widths, BitWidths::Global(7))); + Ok(()) +} + +#[rstest] +#[case::equal_steps(buffer![128u64, 640, 1152].into_array())] +#[case::different_widths(buffer![128u64, 512, 1152].into_array())] +fn into_parts_preserves_block_offsets(#[case] offsets: ArrayRef) -> VortexResult<()> { + let array = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(ByteBuffer::zeroed(1024)), + PType::U32, + Validity::NonNullable, + None, + offsets.clone(), + 1500, + 17, + )?; + let parts = BitPacked::into_parts(array); + let BitWidths::Blocked(block_offsets) = parts.bit_widths else { + vortex_bail!("expected block offsets"); + }; + assert!(ArrayRef::ptr_eq(&offsets, &block_offsets)); + let rebuilt = BitPacked::try_new_with_block_offsets( + parts.packed, + PType::U32, + parts.validity, + parts.patches, + block_offsets, + parts.len, + parts.offset, + )?; + assert_eq!(rebuilt.len(), 1500); + assert_eq!(rebuilt.offset(), 17); + Ok(()) +} diff --git a/encodings/fastlanes/src/bitpacking/array/unpack_iter.rs b/encodings/fastlanes/src/bitpacking/array/unpack_iter.rs index 4877fa9c57f..28724c972a6 100644 --- a/encodings/fastlanes/src/bitpacking/array/unpack_iter.rs +++ b/encodings/fastlanes/src/bitpacking/array/unpack_iter.rs @@ -12,6 +12,7 @@ use lending_iterator::prelude::Item; use lending_iterator::prelude::LendingIterator; use vortex_array::dtype::PhysicalPType; use vortex_error::VortexResult; +use vortex_error::vortex_bail; use vortex_error::vortex_ensure; use crate::BitPackedData; @@ -106,10 +107,13 @@ impl<'a, T: BitPacked> BitUnpackedChunks<'a, T> { len: usize, scratch: &'a mut [MaybeUninit; CHUNK_SIZE], ) -> VortexResult { + let Some(bit_width) = array.global_bit_width else { + vortex_bail!("BitPacked array has per-block bit widths"); + }; Self::try_new_with_strategy( BitPackingStrategy, array.packed_slice::(), - array.bit_width() as usize, + bit_width as usize, array.offset() as usize, len, scratch, diff --git a/encodings/fastlanes/src/bitpacking/compute/between.rs b/encodings/fastlanes/src/bitpacking/compute/between.rs index 8010fa208d2..2a96d484303 100644 --- a/encodings/fastlanes/src/bitpacking/compute/between.rs +++ b/encodings/fastlanes/src/bitpacking/compute/between.rs @@ -22,6 +22,7 @@ use vortex_error::VortexExpect; use vortex_error::VortexResult; use crate::BitPacked; +use crate::BitPackedArrayExt; use crate::bitpacking::compute::stream_predicate::stream_predicate; impl BetweenKernel for BitPacked { @@ -32,6 +33,9 @@ impl BetweenKernel for BitPacked { options: &BetweenOptions, ctx: &mut ExecutionCtx, ) -> VortexResult> { + if !array.bit_widths().is_global() { + return Ok(None); + } // Only accelerate constant-bounds between; vary-by-row bounds fall through to the // default `compare + and` pipeline. let (Some(lower_const), Some(upper_const)) = (lower.as_constant(), upper.as_constant()) diff --git a/encodings/fastlanes/src/bitpacking/compute/cast.rs b/encodings/fastlanes/src/bitpacking/compute/cast.rs index cdbb8141cea..217615be8d6 100644 --- a/encodings/fastlanes/src/bitpacking/compute/cast.rs +++ b/encodings/fastlanes/src/bitpacking/compute/cast.rs @@ -17,6 +17,7 @@ use vortex_array::validity::Validity; use vortex_error::VortexResult; use crate::bitpacking::BitPacked; +use crate::bitpacking::BitWidths; use crate::bitpacking::array::BitPackedArrayExt; use crate::bitpacking::array::bitpack_decompress::unpack_map_into_builder; @@ -33,6 +34,7 @@ fn is_widening_int_cast(src: PType, tgt: PType) -> bool { fn build_with_validity( array: ArrayView<'_, BitPacked>, + bit_width: u8, dtype: &DType, new_validity: Validity, ) -> VortexResult { @@ -44,7 +46,7 @@ fn build_with_validity( .patches() .map(|patches| patches.map_values(|values| values.cast(dtype.clone()))) .transpose()?, - array.bit_width(), + bit_width, array.len(), array.offset(), )? @@ -53,6 +55,9 @@ fn build_with_validity( impl CastReduce for BitPacked { fn cast(array: ArrayView<'_, Self>, dtype: &DType) -> VortexResult> { + let BitWidths::Global(bit_width) = array.bit_widths() else { + return Ok(None); + }; if !array.dtype().eq_ignore_nullability(dtype) { return Ok(None); } @@ -62,7 +67,7 @@ impl CastReduce for BitPacked { else { return Ok(None); }; - build_with_validity(array, dtype, new_validity).map(Some) + build_with_validity(array, bit_width, dtype, new_validity).map(Some) } } @@ -72,13 +77,16 @@ impl CastKernel for BitPacked { dtype: &DType, ctx: &mut ExecutionCtx, ) -> VortexResult> { + let BitWidths::Global(bit_width) = array.bit_widths() else { + return Ok(None); + }; // Nullability-only change: keep the values bit-packed, just adjust validity. if array.dtype().eq_ignore_nullability(dtype) { let new_validity = array .validity()? .cast_nullability(dtype.nullability(), array.len(), ctx)?; - return build_with_validity(array, dtype, new_validity).map(Some); + return build_with_validity(array, bit_width, dtype, new_validity).map(Some); } // Widening integer cast: unpack each FastLanes chunk into a cache-resident scratch buffer diff --git a/encodings/fastlanes/src/bitpacking/compute/compare.rs b/encodings/fastlanes/src/bitpacking/compute/compare.rs index c9d6b815b0d..3eb5b5034d7 100644 --- a/encodings/fastlanes/src/bitpacking/compute/compare.rs +++ b/encodings/fastlanes/src/bitpacking/compute/compare.rs @@ -27,6 +27,8 @@ use vortex_error::VortexExpect; use vortex_error::VortexResult; use crate::BitPacked; +use crate::BitPackedArrayExt; +use crate::BitWidths; use crate::bitpacking::compute::compare_fused::stream_compare_fused; use crate::unpack_iter::BitPacked as BitPackedIter; @@ -37,6 +39,9 @@ impl CompareKernel for BitPacked { operator: CompareOperator, ctx: &mut ExecutionCtx, ) -> VortexResult> { + let BitWidths::Global(bit_width) = lhs.bit_widths() else { + return Ok(None); + }; // Only accelerate compare-against-constant. let Some(constant) = rhs.as_constant() else { return Ok(None); @@ -57,7 +62,7 @@ impl CompareKernel for BitPacked { let rhs: T = constant_prim .typed_value::() .vortex_expect("compare adaptor strips null constants"); - compare_constant_typed::(lhs, rhs, operator, nullability, ctx)? + compare_constant_typed::(lhs, bit_width, rhs, operator, nullability, ctx)? }); Ok(Some(result)) } @@ -69,6 +74,7 @@ impl CompareKernel for BitPacked { /// kernel's dispatch shape). `NotEq` has no direct method, so use `!is_eq`. fn compare_constant_typed( lhs: ArrayView<'_, BitPacked>, + bit_width: u8, rhs: T, operator: CompareOperator, nullability: Nullability, @@ -82,22 +88,22 @@ where { match operator { CompareOperator::Eq => { - stream_compare_fused::(lhs, rhs, nullability, |a, b| a.is_eq(b), ctx) + stream_compare_fused::(lhs, bit_width, rhs, nullability, |a, b| a.is_eq(b), ctx) } CompareOperator::NotEq => { - stream_compare_fused::(lhs, rhs, nullability, |a, b| !a.is_eq(b), ctx) + stream_compare_fused::(lhs, bit_width, rhs, nullability, |a, b| !a.is_eq(b), ctx) } CompareOperator::Lt => { - stream_compare_fused::(lhs, rhs, nullability, |a, b| a.is_lt(b), ctx) + stream_compare_fused::(lhs, bit_width, rhs, nullability, |a, b| a.is_lt(b), ctx) } CompareOperator::Lte => { - stream_compare_fused::(lhs, rhs, nullability, |a, b| a.is_le(b), ctx) + stream_compare_fused::(lhs, bit_width, rhs, nullability, |a, b| a.is_le(b), ctx) } CompareOperator::Gt => { - stream_compare_fused::(lhs, rhs, nullability, |a, b| a.is_gt(b), ctx) + stream_compare_fused::(lhs, bit_width, rhs, nullability, |a, b| a.is_gt(b), ctx) } CompareOperator::Gte => { - stream_compare_fused::(lhs, rhs, nullability, |a, b| a.is_ge(b), ctx) + stream_compare_fused::(lhs, bit_width, rhs, nullability, |a, b| a.is_ge(b), ctx) } } } diff --git a/encodings/fastlanes/src/bitpacking/compute/compare_fused.rs b/encodings/fastlanes/src/bitpacking/compute/compare_fused.rs index 1259ed815fe..ecc1be33de8 100644 --- a/encodings/fastlanes/src/bitpacking/compute/compare_fused.rs +++ b/encodings/fastlanes/src/bitpacking/compute/compare_fused.rs @@ -65,6 +65,7 @@ const WORDS_PER_CHUNK: usize = CHUNK_SIZE / U64_BITS; /// [`BitPackedArray`]: crate::BitPackedArray pub(super) fn stream_compare_fused( array: ArrayView<'_, BitPacked>, + bit_width: u8, rhs: T, nullability: Nullability, cmp: F, @@ -78,7 +79,7 @@ where F: Fn(T, T) -> bool + Copy, { let len = array.len(); - let bit_width = array.bit_width() as usize; + let bit_width = bit_width as usize; let offset = array.offset() as usize; // A degenerate width has no packed payload for the fused kernel to consume; defer to the scalar diff --git a/encodings/fastlanes/src/bitpacking/compute/filter.rs b/encodings/fastlanes/src/bitpacking/compute/filter.rs index 0b1b9422f86..330c60b9e3b 100644 --- a/encodings/fastlanes/src/bitpacking/compute/filter.rs +++ b/encodings/fastlanes/src/bitpacking/compute/filter.rs @@ -26,6 +26,7 @@ use super::take::UNPACK_CHUNK_THRESHOLD; use crate::BitPacked; use crate::BitPackedArrayExt; use crate::BitPackedData; +use crate::BitWidths; /// The threshold over which it is faster to fully unpack the entire [`BitPackedArray`](crate::BitPackedArray) and then /// filter the result than to unpack only specific bitpacked values into the output buffer. @@ -49,6 +50,9 @@ impl FilterKernel for BitPacked { mask: &Mask, ctx: &mut ExecutionCtx, ) -> VortexResult> { + let BitWidths::Global(bit_width) = array.bit_widths() else { + return Ok(None); + }; let values = match mask { Mask::AllTrue(_) | Mask::AllFalse(_) => { return Ok(None); @@ -65,7 +69,8 @@ impl FilterKernel for BitPacked { // Filter and patch using the correct unsigned type for FastLanes, then cast to signed if needed. let primitive = match_each_unsigned_integer_ptype!(array.dtype().as_ptype().to_unsigned(), |U| { - let (buffer, validity) = filter_primitive_without_patches::(array, values)?; + let (buffer, validity) = + filter_primitive_without_patches::(array, bit_width, values)?; // reinterpret_cast for signed types. let primitive = PrimitiveArray::new(buffer, validity); if array.dtype().as_ptype().is_signed_int() { @@ -108,9 +113,10 @@ impl FilterKernel for BitPacked { /// Returns a tuple of (values buffer, validity mask). fn filter_primitive_without_patches( array: ArrayView<'_, BitPacked>, + bit_width: u8, selection: &MaskValuesRef, ) -> VortexResult<(Buffer, Validity)> { - let values = filter_with_indices(array.data(), selection.indices()); + let values = filter_with_indices(array.data(), bit_width, selection.indices()); let validity = array .validity()? .filter(&Mask::Values(MaskValuesRef::clone(selection)))?; @@ -120,10 +126,11 @@ fn filter_primitive_without_patches( fn filter_with_indices( array: &BitPackedData, + bit_width: u8, indices: &[usize], ) -> BufferMut { let offset = array.offset() as usize; - let bit_width = array.bit_width() as usize; + let bit_width = bit_width as usize; let mut values = BufferMut::with_capacity(indices.len()); // Some re-usable memory to store per-chunk indices. diff --git a/encodings/fastlanes/src/bitpacking/compute/is_constant.rs b/encodings/fastlanes/src/bitpacking/compute/is_constant.rs index 0ab01a635ba..0ad8005ba0f 100644 --- a/encodings/fastlanes/src/bitpacking/compute/is_constant.rs +++ b/encodings/fastlanes/src/bitpacking/compute/is_constant.rs @@ -43,6 +43,9 @@ impl DynAggregateKernel for BitPackedIsConstantKernel { let Some(array) = batch.as_opt::() else { return Ok(None); }; + if !array.bit_widths().is_global() { + return Ok(None); + } let result = match_each_integer_ptype!(array.dtype().as_ptype(), |P| { bitpacked_is_constant::() }>(array, ctx)? diff --git a/encodings/fastlanes/src/bitpacking/compute/slice.rs b/encodings/fastlanes/src/bitpacking/compute/slice.rs index 996565a2672..3175bec2a57 100644 --- a/encodings/fastlanes/src/bitpacking/compute/slice.rs +++ b/encodings/fastlanes/src/bitpacking/compute/slice.rs @@ -12,18 +12,25 @@ use vortex_array::arrays::slice::SliceKernel; use vortex_array::arrays::slice::SliceReduce; use vortex_array::patches::Patches; use vortex_error::VortexResult; +use vortex_error::vortex_ensure; use crate::BitPacked; +use crate::BitWidths; +use crate::FL_CHUNK_SIZE; use crate::bitpacking::array::BitPackedArrayExt; impl SliceReduce for BitPacked { fn slice(array: ArrayView<'_, Self>, range: Range) -> VortexResult> { + // Slicing per-block bit widths reads block boundaries, which needs the kernel. + let BitWidths::Global(bit_width) = array.bit_widths() else { + return Ok(None); + }; // We cannot access buffers (to slice the patches). if array.patches().is_some() { return Ok(None); } - Ok(Some(slice_bitpacked(array, range, None)?)) + Ok(Some(slice_bitpacked(array, bit_width, range, None)?)) } } @@ -31,7 +38,7 @@ impl SliceKernel for BitPacked { fn slice( array: ArrayView<'_, Self>, range: Range, - _ctx: &mut ExecutionCtx, + ctx: &mut ExecutionCtx, ) -> VortexResult> { let patches = array .patches() @@ -39,12 +46,18 @@ impl SliceKernel for BitPacked { .transpose()? .flatten(); - Ok(Some(slice_bitpacked(array, range, patches)?)) + Ok(Some(match array.bit_widths() { + BitWidths::Global(bit_width) => slice_bitpacked(array, bit_width, range, patches)?, + BitWidths::Blocked(block_offsets) => { + slice_blocked(array, &block_offsets, range, patches, ctx)? + } + })) } } fn slice_bitpacked( array: ArrayView<'_, BitPacked>, + bit_width: u8, range: Range, patches: Option, ) -> VortexResult { @@ -54,32 +67,93 @@ fn slice_bitpacked( let block_start = max(0, offset_start - offset); let block_stop = offset_stop.div_ceil(1024) * 1024; - let encoded_start = (block_start / 8) * array.bit_width() as usize; - let encoded_stop = (block_stop / 8) * array.bit_width() as usize; + let encoded_start = (block_start / 8) * bit_width as usize; + let encoded_stop = (block_stop / 8) * bit_width as usize; Ok(BitPacked::try_new( array.packed().slice(encoded_start..encoded_stop), array.dtype().as_ptype(), array.validity()?.slice(range.clone())?, patches, - array.bit_width(), + bit_width, range.len(), offset as u16, )? .into_array()) } +/// Keep the blocks that `range` overlaps. +/// +/// Block boundaries are relative to the first one, so the overlapped blocks' boundaries are the +/// new block offsets as they are, and the packed buffer is cut between the first and last of them. +/// +/// E.g. rows 1500..4100 of an array with block offsets `[b0, b1, b2, b3, b4, b5]` keep blocks 1 to +/// 4: the block offsets become `[b1, b2, b3, b4, b5]`, the packed buffer is cut to bytes +/// `b1 - b0..b5 - b0`, and the offset is 476 (row 1500 is 476 rows into block 1). +fn slice_blocked( + array: ArrayView<'_, BitPacked>, + block_offsets: &ArrayRef, + range: Range, + patches: Option, + ctx: &mut ExecutionCtx, +) -> VortexResult { + let offset_start = range.start + array.offset() as usize; + let offset_stop = range.end + array.offset() as usize; + let first_block = offset_start / FL_CHUNK_SIZE; + let end_block = offset_stop.div_ceil(FL_CHUNK_SIZE); + + let mut boundary = |block: usize| u64::try_from(&block_offsets.execute_scalar(block, ctx)?); + let base = boundary(0)?; + let start = boundary(first_block)?; + let end = boundary(end_block)?; + // Release builds don't validate boundaries on construction, so check them before slicing. + vortex_ensure!( + base <= start && start <= end && end - base <= array.packed().len() as u64, + "Block boundaries {start} and {end} are outside the packed buffer (base {base})" + ); + + // Both differences are at most the packed length, so they fit in `usize`. + let packed = array + .packed() + .slice((start - base) as usize..(end - base) as usize); + Ok(BitPacked::try_new_with_block_offsets( + packed, + array.dtype().as_ptype(), + array.validity()?.slice(range.clone())?, + patches, + block_offsets.slice(first_block..end_block + 1)?, + range.len(), + (offset_start % FL_CHUNK_SIZE) as u16, + )? + .into_array()) +} + #[cfg(test)] mod tests { + use std::ops::Range; + + use rstest::rstest; + use vortex_array::ArrayRef; use vortex_array::IntoArray; use vortex_array::VortexSessionExecute; use vortex_array::array_session; use vortex_array::arrays::PrimitiveArray; use vortex_array::arrays::SliceArray; + use vortex_array::arrays::slice::SliceKernel; + use vortex_array::assert_arrays_eq; + use vortex_array::scalar::Scalar; use vortex_error::VortexResult; + use vortex_error::vortex_bail; + use vortex_error::vortex_err; use crate::BitPacked; + use crate::BitPackedArray; + use crate::BitPackedArrayExt; + use crate::BitWidths; + use crate::FoR; use crate::bitpack_compress::bitpack_encode; + use crate::bitpack_compress::bitpack_encode_blocked; + use crate::test::SESSION; #[test] fn test_reduce_parent_returns_bitpacked_slice() -> VortexResult<()> { @@ -101,4 +175,131 @@ mod tests { Ok(()) } + + /// Values whose 1024-value blocks need 2 to 6 bits. + fn drifting() -> PrimitiveArray { + PrimitiveArray::from_iter((0..5000u32).map(|i| i % (4 << (i / 1024)))) + } + + /// Pack `values` at `bit_widths`, with FoR-encoded block offsets if `encoded_offsets`. + fn blocked( + values: &PrimitiveArray, + bit_widths: &[u8], + encoded_offsets: bool, + ) -> VortexResult { + let array = bitpack_encode_blocked( + values, + bit_widths, + None, + &mut SESSION.create_execution_ctx(), + )?; + if !encoded_offsets { + return Ok(array); + } + let BitWidths::Blocked(block_offsets) = array.bit_widths() else { + vortex_bail!("expected block offsets"); + }; + BitPacked::try_new_with_block_offsets( + array.packed().clone(), + array.dtype().as_ptype(), + array.validity()?, + array.patches(), + FoR::try_new( + block_offsets.clone(), + Scalar::zero_value(block_offsets.dtype()), + )? + .into_array(), + array.len(), + array.offset(), + ) + } + + /// Slice `array` with the kernel, which must keep it blocked. + fn slice_kernel(array: &ArrayRef, range: Range) -> VortexResult { + let sliced = ::slice( + array.as_::(), + range, + &mut SESSION.create_execution_ctx(), + )? + .ok_or_else(|| vortex_err!("expected the slice kernel to slice"))?; + let sliced: BitPackedArray = sliced + .try_downcast() + .map_err(|a| vortex_err!("expected BitPacked, got {}", a.encoding_id()))?; + assert!(!sliced.bit_widths().is_global()); + Ok(sliced) + } + + #[rstest] + #[case::whole(0..5000)] + #[case::within_block(1100..1900)] + #[case::across_blocks(1500..4100)] + #[case::block_aligned(1024..3072)] + #[case::to_end(3000..5000)] + #[case::empty_at_boundary(2048..2048)] + #[case::empty_within_block(2100..2100)] + fn slice_blocked( + #[case] range: Range, + #[values(false, true)] encoded_offsets: bool, + ) -> VortexResult<()> { + let values = drifting(); + let array = blocked(&values, &[2, 3, 4, 5, 6], encoded_offsets)?.into_array(); + let sliced = slice_kernel(&array, range.clone())?; + assert_arrays_eq!( + sliced, + values.into_array().slice(range)?, + &mut SESSION.create_execution_ctx() + ); + Ok(()) + } + + #[test] + fn slice_blocked_keeps_only_the_overlapped_blocks() -> VortexResult<()> { + let array = blocked(&drifting(), &[2, 3, 4, 5, 6], false)?.into_array(); + // Rows 1500..4100 overlap blocks 1 to 4, packed at 3, 4, 5 and 6 bits. + let sliced = slice_kernel(&array, 1500..4100)?; + assert_eq!(sliced.offset(), 476); + assert_eq!(sliced.packed().len(), 128 * (3 + 4 + 5 + 6)); + let BitWidths::Blocked(block_offsets) = sliced.bit_widths() else { + vortex_bail!("expected block offsets"); + }; + assert_arrays_eq!( + block_offsets, + PrimitiveArray::from_iter([256u16, 640, 1152, 1792, 2560]), + &mut SESSION.create_execution_ctx() + ); + Ok(()) + } + + #[test] + fn slice_blocked_with_nulls_and_patches() -> VortexResult<()> { + let values = PrimitiveArray::from_option_iter( + (0..5000u32).map(|i| (i % 7 != 0).then_some(i % (4 << (i / 1024)))), + ); + // The last block packs values up to 63 at 3 bits, so it has patches. + let array = blocked(&values, &[2, 3, 4, 5, 3], false)?.into_array(); + assert!(array.as_::().patches().is_some()); + let sliced = slice_kernel(&array, 3000..4900)?; + assert!(sliced.patches().is_some()); + assert_arrays_eq!( + sliced, + values.into_array().slice(3000..4900)?, + &mut SESSION.create_execution_ctx() + ); + Ok(()) + } + + #[test] + fn slice_blocked_twice() -> VortexResult<()> { + let values = drifting(); + let array = blocked(&values, &[2, 3, 4, 5, 6], true)?.into_array(); + let once = slice_kernel(&array, 1500..4900)?.into_array(); + let twice = slice_kernel(&once, 300..2000)?; + assert_eq!(twice.offset(), 776); + assert_arrays_eq!( + twice, + values.into_array().slice(1800..3500)?, + &mut SESSION.create_execution_ctx() + ); + Ok(()) + } } diff --git a/encodings/fastlanes/src/bitpacking/compute/take.rs b/encodings/fastlanes/src/bitpacking/compute/take.rs index 86e97623cf6..c3d8b54a7ab 100644 --- a/encodings/fastlanes/src/bitpacking/compute/take.rs +++ b/encodings/fastlanes/src/bitpacking/compute/take.rs @@ -25,6 +25,7 @@ use vortex_error::VortexResult; use super::chunked_indices; use crate::BitPacked; use crate::BitPackedArrayExt; +use crate::BitWidths; use crate::bitpack_decompress; // TODO(connor): This is duplicated in `encodings/fastlanes/src/bitpacking/kernels/mod.rs`. @@ -39,6 +40,9 @@ impl TakeExecute for BitPacked { indices: &ArrayRef, ctx: &mut ExecutionCtx, ) -> VortexResult> { + let BitWidths::Global(bit_width) = array.bit_widths() else { + return Ok(None); + }; // If the indices are large enough, it's faster to flatten and take the primitive array. if indices.len() * UNPACK_CHUNK_THRESHOLD > array.len() { let prim = array.array().clone().execute::(ctx)?; @@ -54,7 +58,7 @@ impl TakeExecute for BitPacked { let indices = indices.clone().execute::(ctx)?; let taken = match_each_unsigned_integer_ptype!(ptype.to_unsigned(), |T| { match_each_integer_ptype!(indices.ptype(), |I| { - take_primitive::(array, &indices, taken_validity, ctx)? + take_primitive::(array, bit_width, &indices, taken_validity, ctx)? }) }); let taken = if ptype.is_signed_int() { @@ -72,6 +76,7 @@ impl TakeExecute for BitPacked { fn take_primitive( array: ArrayView<'_, BitPacked>, + bit_width: u8, indices: &PrimitiveArray, taken_validity: Validity, ctx: &mut ExecutionCtx, @@ -81,7 +86,7 @@ fn take_primitive( } let offset = array.offset() as usize; - let bit_width = array.bit_width() as usize; + let bit_width = bit_width as usize; let packed = array.packed_slice::(); @@ -279,6 +284,7 @@ mod test { let taken_primitive = take_primitive::( start.as_view(), + 1, &PrimitiveArray::from_iter([0u64, 1, 2, 3]), Validity::NonNullable, &mut ctx, diff --git a/encodings/fastlanes/src/bitpacking/mod.rs b/encodings/fastlanes/src/bitpacking/mod.rs index f6af27bf728..5233ab64e07 100644 --- a/encodings/fastlanes/src/bitpacking/mod.rs +++ b/encodings/fastlanes/src/bitpacking/mod.rs @@ -7,6 +7,7 @@ pub use array::BitPackedArraySlotsExt; pub use array::BitPackedData; pub use array::BitPackedDataParts; pub use array::BitPackedSlots; +pub use array::BitWidths; pub use array::bitpack_compress; pub use array::bitpack_decompress; pub use array::unpack_iter; @@ -18,6 +19,8 @@ mod vtable; pub(crate) use plugin::BitPackedPatchedPlugin; pub use plugin::BitPackedPlugin; +pub use plugin::bitpacked_v1_id; +pub use plugin::bitpacked_v2_id; pub use vtable::BitPacked; pub use vtable::BitPackedArray; diff --git a/encodings/fastlanes/src/bitpacking/plugin.rs b/encodings/fastlanes/src/bitpacking/plugin.rs deleted file mode 100644 index aa8cc97c946..00000000000 --- a/encodings/fastlanes/src/bitpacking/plugin.rs +++ /dev/null @@ -1,482 +0,0 @@ -// SPDX-License-Identifier: Apache-2.0 -// SPDX-FileCopyrightText: Copyright the Vortex contributors - -//! [`ArrayPlugin`] implementations for `BitPacked`. -//! -//! [`BitPackedPlugin`] owns serde for `fastlanes.bitpacked`. [`BitPackedPatchedPlugin`] lets you -//! load in and deserialize a `BitPacked` array with interior patches as a `PatchedArray` that wraps -//! a patchless `BitPacked` array. -//! -//! This enables zero-cost backward compatibility with previously written datasets. - -use prost::Message; -use vortex_array::Array; -use vortex_array::ArrayDeserialization; -use vortex_array::ArrayId; -use vortex_array::ArrayParts; -use vortex_array::ArrayPlugin; -use vortex_array::ArrayRef; -use vortex_array::ArraySerialization; -use vortex_array::ArraySlots; -use vortex_array::ArrayVTable; -use vortex_array::IntoArray; -use vortex_array::VortexSessionExecute; -use vortex_array::arrays::Patched; -use vortex_array::patches::Patches; -use vortex_array::patches::PatchesData; -use vortex_array::patches::PatchesMetadata; -use vortex_array::validity::Validity; -use vortex_array::vtable::validity_to_child; -use vortex_error::VortexResult; -use vortex_error::vortex_bail; -use vortex_error::vortex_ensure; -use vortex_error::vortex_err; -use vortex_session::VortexSession; - -use crate::BitPacked; -use crate::BitPackedArray; -use crate::BitPackedArrayExt; -use crate::BitPackedData; - -#[derive(Clone, prost::Message)] -pub struct BitPackedMetadata { - #[prost(uint32, tag = "1")] - pub(crate) bit_width: u32, - #[prost(uint32, tag = "2")] - pub(crate) offset: u32, // must be <1024 - #[prost(message, optional, tag = "3")] - pub(crate) patches: Option, -} - -/// Serde for the [`BitPacked`] array. -/// -/// Register this plugin, or call [`crate::initialize`], to enable serde. Direct registration of -/// [`BitPacked`] does not support serde. -#[derive(Clone, Debug)] -pub struct BitPackedPlugin; - -impl ArrayPlugin for BitPackedPlugin { - fn id(&self) -> ArrayId { - ArrayVTable::id(&BitPacked) - } - - fn serialize( - &self, - array: &ArrayRef, - _session: &VortexSession, - ) -> VortexResult> { - let view = array.as_opt::().ok_or_else(|| { - vortex_err!("BitPacked plugin cannot serialize {}", array.encoding_id()) - })?; - let metadata = BitPackedMetadata { - bit_width: view.bit_width() as u32, - offset: view.offset() as u32, - patches: view - .patches() - .map(|p| p.to_metadata(view.len(), view.dtype())) - .transpose()?, - } - .encode_to_vec(); - Ok(Some(ArraySerialization::from_array( - self.id(), - array, - metadata, - ))) - } - - fn deserialize( - &self, - parts: ArrayDeserialization<'_>, - _session: &VortexSession, - ) -> VortexResult { - vortex_ensure!( - parts.serialized_id == self.id(), - "BitPacked plugin does not recognize serialized ID {}", - parts.serialized_id - ); - let ArrayDeserialization { - dtype, - len, - metadata, - buffers, - children, - .. - } = parts; - - let metadata = BitPackedMetadata::decode(metadata)?; - if buffers.len() != 1 { - vortex_bail!("Expected 1 buffer, got {}", buffers.len()); - } - let packed = buffers[0].clone(); - - let load_validity = |child_idx: usize| { - if children.len() == child_idx { - Ok(Validity::from(dtype.nullability())) - } else if children.len() == child_idx + 1 { - let validity = children.get(child_idx, &Validity::DTYPE, len)?; - Ok(Validity::Array(validity)) - } else { - vortex_bail!( - "Expected {} or {} children, got {}", - child_idx, - child_idx + 1, - children.len() - ); - } - }; - - let validity_idx = match &metadata.patches { - None => 0, - Some(patches_meta) if patches_meta.chunk_offsets_dtype()?.is_some() => 3, - Some(_) => 2, - }; - - let validity = load_validity(validity_idx)?; - - let patches = metadata - .patches - .map(|p| { - let indices = children.get(0, &p.indices_dtype()?, p.len()?)?; - let values = children.get(1, dtype, p.len()?)?; - let chunk_offsets = p - .chunk_offsets_dtype()? - .map(|dtype| children.get(2, &dtype, p.chunk_offsets_len() as usize)) - .transpose()?; - - Patches::new(len, p.offset()?, indices, values, chunk_offsets) - }) - .transpose()?; - - let slots = { - let mut s = ArraySlots::with_capacity(4); - PatchesData::push_slots(&mut s, patches.as_ref()); - s.push(validity_to_child(&validity, len)); - s - }; - let data = BitPackedData::try_new( - packed, - patches, - u8::try_from(metadata.bit_width).map_err(|_| { - vortex_err!( - "BitPackedMetadata bit_width {} does not fit in u8", - metadata.bit_width - ) - })?, - u16::try_from(metadata.offset).map_err(|_| { - vortex_err!( - "BitPackedMetadata offset {} does not fit in u16", - metadata.offset - ) - })?, - )?; - Ok(Array::::try_from_parts( - ArrayParts::new(BitPacked, dtype.clone(), len, data).with_slots(slots), - )? - .into_array()) - } -} - -/// Custom deserialization plugin that converts a BitPacked array with interior -/// Patches into a PatchedArray holding a BitPacked array. -#[derive(Debug, Clone)] -pub(crate) struct BitPackedPatchedPlugin; - -impl ArrayPlugin for BitPackedPatchedPlugin { - fn id(&self) -> ArrayId { - // We reuse the existing `BitPacked` ID so that we can take over its - // deserialization pathway. - // TODO(joe): dedup method name - ArrayVTable::id(&BitPacked) - } - - fn serialize( - &self, - array: &ArrayRef, - session: &VortexSession, - ) -> VortexResult> { - // delegate to BitPackedPlugin for serialization - BitPackedPlugin.serialize(array, session) - } - - fn deserialize( - &self, - parts: ArrayDeserialization<'_>, - session: &VortexSession, - ) -> VortexResult { - vortex_ensure!( - parts.serialized_id == self.id(), - "BitPacked plugin does not recognize serialized ID {}", - parts.serialized_id, - ); - let bitpacked: BitPackedArray = BitPackedPlugin - .deserialize(parts, session)? - .try_downcast() - .map_err(|_| { - vortex_err!("BitPacked plugin should only deserialize fastlanes.bitpacked") - })?; - - // Create a new BitPackedArray without the interior patches installed. - let Some(patches) = bitpacked.patches() else { - return Ok(bitpacked.into_array()); - }; - - let packed = bitpacked.packed().clone(); - let ptype = bitpacked.dtype().as_ptype(); - let validity = bitpacked.validity()?; - let bw = bitpacked.bit_width; - let len = bitpacked.len(); - let offset = bitpacked.offset(); - - let bitpacked_without_patches = - BitPacked::try_new(packed, ptype, validity, None, bw, len, offset)?.into_array(); - - let patched = Patched::from_array_and_patches( - bitpacked_without_patches, - &patches, - &mut session.create_execution_ctx(), - )?; - - Ok(patched.into_array()) - } - - fn is_supported_encoding(&self, id: &ArrayId) -> bool { - id == ArrayVTable::id(&BitPacked) || id == ArrayVTable::id(&Patched) - } -} - -#[cfg(test)] -mod tests { - use std::sync::LazyLock; - - use prost::Message; - use rstest::rstest; - use vortex_array::ArrayContext; - use vortex_array::ArrayDeserialization; - use vortex_array::ArrayPlugin; - use vortex_array::ArrayRef; - use vortex_array::IntoArray; - use vortex_array::VortexSessionExecute; - use vortex_array::arrays::PatchedArray; - use vortex_array::arrays::PrimitiveArray; - use vortex_array::arrays::patched::PatchedArraySlotsExt; - use vortex_array::assert_arrays_eq; - use vortex_array::buffer::BufferHandle; - use vortex_array::serde::SerializeOptions; - use vortex_array::serde::SerializedArray; - use vortex_array::session::ArraySessionExt; - use vortex_buffer::Buffer; - use vortex_buffer::ByteBufferMut; - use vortex_error::VortexResult; - use vortex_error::vortex_err; - use vortex_session::VortexSession; - use vortex_session::registry::ReadContext; - - use super::BitPackedMetadata; - use super::BitPackedPatchedPlugin; - use super::BitPackedPlugin; - use crate::BitPacked; - use crate::BitPackedArray; - use crate::BitPackedArrayExt; - use crate::BitPackedData; - - static SESSION: LazyLock = LazyLock::new(|| { - let session = vortex_array::array_session(); - session.arrays().register(BitPackedPatchedPlugin); - session - }); - - #[test] - fn test_decode_bitpacked_patches() -> VortexResult<()> { - let mut ctx = SESSION.create_execution_ctx(); - // Create values where some exceed the bit width, causing patches. - // With bit_width=9, max value is 511. Values >=512 become patches. - let values: Buffer = (0i32..=512).collect(); - let parray = values.into_array(); - let bitpacked = BitPackedData::encode(&parray, 9, &mut ctx)?; - - assert!( - bitpacked.patches().is_some(), - "Expected BitPacked array to have patches" - ); - - let array = bitpacked.as_array(); - - let serialization = SESSION.array_serialize(array)?.unwrap(); - let children = array.children(); - let buffers = array - .buffers() - .into_iter() - .map(BufferHandle::new_host) - .collect::>(); - - let deserialized = BitPackedPatchedPlugin.deserialize( - ArrayDeserialization::new( - BitPackedPatchedPlugin.id(), - array.dtype(), - array.len(), - &serialization.metadata, - &buffers, - &children, - ), - &SESSION, - )?; - - let patched: PatchedArray = deserialized - .try_downcast() - .map_err(|a| vortex_err!("Expected Patched, got {}", a.encoding_id()))?; - - let inner_bitpacked: BitPackedArray = patched - .inner() - .clone() - .try_downcast() - .map_err(|a| vortex_err!("Expected inner BitPacked, got {}", a.encoding_id()))?; - - assert!( - inner_bitpacked.patches().is_none(), - "Inner BitPacked should NOT have patches" - ); - - Ok(()) - } - - #[test] - fn bitpacked_without_patches_stays_bitpacked() -> VortexResult<()> { - let mut ctx = SESSION.create_execution_ctx(); - // With bit_width=16, max value is 65535. All values 0..100 fit. - let values: Buffer = (0i32..100).collect(); - let parray = values.into_array(); - let bitpacked = BitPackedData::encode(&parray, 16, &mut ctx)?; - - assert!( - bitpacked.patches().is_none(), - "Expected BitPacked array without patches" - ); - - let array = bitpacked.as_array(); - - let serialization = SESSION.array_serialize(array)?.unwrap(); - let children = array.children(); - let buffers = array - .buffers() - .into_iter() - .map(BufferHandle::new_host) - .collect::>(); - - let deserialized = BitPackedPatchedPlugin.deserialize( - ArrayDeserialization::new( - BitPackedPatchedPlugin.id(), - array.dtype(), - array.len(), - &serialization.metadata, - &buffers, - &children, - ), - &SESSION, - )?; - - let result = deserialized - .try_downcast::() - .map_err(|a| vortex_err!("Expected deserialize BitPacked, got {}", a.encoding_id()))?; - - assert!(result.patches().is_none(), "Result should not have patches"); - - Ok(()) - } - - #[test] - fn primitive_array_returns_error() -> VortexResult<()> { - let array = PrimitiveArray::from_iter([1i32, 2, 3]).into_array(); - - let serialization = SESSION.array_serialize(&array)?.unwrap(); - let children = array.children(); - let buffers = array - .buffers() - .into_iter() - .map(BufferHandle::new_host) - .collect::>(); - - let result = BitPackedPatchedPlugin.deserialize( - ArrayDeserialization::new( - BitPackedPatchedPlugin.id(), - array.dtype(), - array.len(), - &serialization.metadata, - &buffers, - &children, - ), - &SESSION, - ); - - assert!( - result.is_err(), - "Expected error when deserializing PrimitiveArray with BitPackedPatchedPlugin" - ); - - Ok(()) - } - - static PLUGIN_SESSION: LazyLock = LazyLock::new(|| { - let session = vortex_array::array_session(); - session.arrays().register(BitPackedPlugin); - session - }); - - fn roundtrip(array: &ArrayRef) -> VortexResult { - let array_ctx = ArrayContext::empty(); - let buffers = array.serialize(&array_ctx, &PLUGIN_SESSION, &SerializeOptions::default())?; - let mut bytes = ByteBufferMut::empty(); - for buffer in buffers { - bytes.extend_from_slice(&buffer); - } - SerializedArray::try_from(bytes.freeze())?.decode( - array.dtype(), - array.len(), - &ReadContext::new(array_ctx.to_ids()), - &PLUGIN_SESSION, - ) - } - - #[rstest] - #[case::no_patches(PrimitiveArray::from_iter(0u32..3000).into_array(), 12, 0..3000)] - #[case::patches(PrimitiveArray::from_iter(0i32..=2048).into_array(), 9, 0..2049)] - #[case::nullable( - PrimitiveArray::from_option_iter((0u16..3000).map(|i| (i % 5 != 0).then_some(i))).into_array(), - 8, - 0..3000, - )] - #[case::sliced(PrimitiveArray::from_iter(0u32..3000).into_array(), 12, 700..1900)] - fn serde_roundtrip( - #[case] values: ArrayRef, - #[case] bit_width: u8, - #[case] range: std::ops::Range, - ) -> VortexResult<()> { - let mut ctx = PLUGIN_SESSION.create_execution_ctx(); - let array = BitPackedData::encode(&values, bit_width, &mut ctx)? - .into_array() - .slice(range.clone())?; - let view = array.as_::(); - let serialization = PLUGIN_SESSION - .array_serialize(&array)? - .ok_or_else(|| vortex_err!("BitPacked must serialize"))?; - assert_eq!(serialization.serialized_id, BitPackedPlugin.id()); - let metadata = BitPackedMetadata::decode(serialization.metadata.as_slice())?; - assert_eq!(metadata.bit_width, u32::from(bit_width)); - assert_eq!(metadata.offset, u32::from(view.offset())); - assert_eq!(metadata.patches.is_some(), view.patches().is_some()); - - let read = roundtrip(&array)?; - assert_eq!(read.encoding_id(), BitPackedPlugin.id()); - assert_arrays_eq!(read, values.slice(range)?, &mut ctx); - Ok(()) - } - - #[test] - fn vtable_serde_requires_plugin() -> VortexResult<()> { - let values = PrimitiveArray::from_iter([1u8, 2, 3]).into_array(); - let array = BitPackedData::encode(&values, 2, &mut SESSION.create_execution_ctx())?; - let session = vortex_array::array_session(); - session.arrays().register(BitPacked); - assert!(session.array_serialize(&array.into_array()).is_err()); - Ok(()) - } -} diff --git a/encodings/fastlanes/src/bitpacking/plugin/mod.rs b/encodings/fastlanes/src/bitpacking/plugin/mod.rs new file mode 100644 index 00000000000..ddcdff77792 --- /dev/null +++ b/encodings/fastlanes/src/bitpacking/plugin/mod.rs @@ -0,0 +1,186 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +//! [`ArrayPlugin`] implementations for `BitPacked`. +//! +//! [`BitPackedPlugin`] owns serde for `fastlanes.bitpacked` and `fastlanes.bitpacked.v2`. +//! [`BitPackedPatchedPlugin`] lets you load in and deserialize a `BitPacked` array with interior +//! patches as a `PatchedArray` that wraps a patchless `BitPacked` array. +//! +//! This enables zero-cost backward compatibility with previously written datasets. + +use vortex_array::ArrayDeserialization; +use vortex_array::ArrayId; +use vortex_array::ArrayPlugin; +use vortex_array::ArrayRef; +use vortex_array::ArraySerialization; +use vortex_array::ArrayVTable; +use vortex_array::IntoArray; +use vortex_array::VortexSessionExecute; +use vortex_array::arrays::Patched; +use vortex_error::VortexResult; +use vortex_error::vortex_bail; +use vortex_error::vortex_ensure; +use vortex_error::vortex_err; +use vortex_session::VortexSession; +use vortex_session::registry::CachedId; + +use crate::BitPacked; +use crate::BitPackedArray; +use crate::BitPackedArrayExt; +use crate::BitWidths; + +#[cfg(test)] +mod tests; +mod v1; +mod v2; + +/// The frozen BitPacked serialized ID for arrays whose blocks share one bit width. +pub fn bitpacked_v1_id() -> ArrayId { + static ID: CachedId = CachedId::new("fastlanes.bitpacked"); + *ID +} + +/// The in-memory BitPacked ID, and the serialized ID for arrays whose blocks have their own bit +/// widths. +pub fn bitpacked_v2_id() -> ArrayId { + static ID: CachedId = CachedId::new("fastlanes.bitpacked.v2"); + *ID +} + +/// Serde for the [`BitPacked`] array. +/// +/// Arrays with a global bit width serialize as `fastlanes.bitpacked`, which stores the width in +/// its metadata. Arrays whose blocks have their own bit widths serialize as +/// `fastlanes.bitpacked.v2`, which stores the block offsets as a child. +/// +/// Register this plugin, or call [`crate::initialize`], to enable serde. Direct registration of +/// [`BitPacked`] does not support serde. +#[derive(Clone, Debug)] +pub struct BitPackedPlugin; + +impl ArrayPlugin for BitPackedPlugin { + fn id(&self) -> ArrayId { + ArrayVTable::id(&BitPacked) + } + + fn serialized_ids(&self) -> Vec { + vec![bitpacked_v1_id(), bitpacked_v2_id()] + } + + fn serialize( + &self, + array: &ArrayRef, + _session: &VortexSession, + ) -> VortexResult> { + let view = array.as_opt::().ok_or_else(|| { + vortex_err!("BitPacked plugin cannot serialize {}", array.encoding_id()) + })?; + let serialization = match view.bit_widths() { + BitWidths::Global(bit_width) => v1::serialize(array, view, bit_width)?, + BitWidths::Blocked(block_offsets) => v2::serialize(array, view, &block_offsets)?, + }; + Ok(Some(serialization)) + } + + fn deserialize( + &self, + parts: ArrayDeserialization<'_>, + _session: &VortexSession, + ) -> VortexResult { + if parts.serialized_id == bitpacked_v1_id() { + v1::deserialize(parts) + } else if parts.serialized_id == bitpacked_v2_id() { + v2::deserialize(parts) + } else { + vortex_bail!( + "BitPacked plugin does not recognize serialized ID {}", + parts.serialized_id + ) + } + } +} + +/// Custom deserialization plugin that converts a BitPacked array with interior +/// Patches into a PatchedArray holding a BitPacked array. +#[derive(Debug, Clone)] +pub(crate) struct BitPackedPatchedPlugin; + +impl ArrayPlugin for BitPackedPatchedPlugin { + fn id(&self) -> ArrayId { + // We reuse the existing `BitPacked` ID so that we can take over its + // deserialization pathway. + // TODO(joe): dedup method name + ArrayVTable::id(&BitPacked) + } + + fn serialized_ids(&self) -> Vec { + BitPackedPlugin.serialized_ids() + } + + fn serialize( + &self, + array: &ArrayRef, + session: &VortexSession, + ) -> VortexResult> { + // delegate to BitPackedPlugin for serialization + BitPackedPlugin.serialize(array, session) + } + + fn deserialize( + &self, + parts: ArrayDeserialization<'_>, + session: &VortexSession, + ) -> VortexResult { + vortex_ensure!( + self.serialized_ids().contains(&parts.serialized_id), + "BitPacked plugin does not recognize serialized ID {}", + parts.serialized_id, + ); + let bitpacked: BitPackedArray = BitPackedPlugin + .deserialize(parts, session)? + .try_downcast() + .map_err(|_| { + vortex_err!("BitPacked plugin should only deserialize fastlanes.bitpacked") + })?; + + // Create a new BitPackedArray without the interior patches installed. + let Some(patches) = bitpacked.patches() else { + return Ok(bitpacked.into_array()); + }; + + let packed = bitpacked.packed().clone(); + let ptype = bitpacked.dtype().as_ptype(); + let validity = bitpacked.validity()?; + let len = bitpacked.len(); + let offset = bitpacked.offset(); + + let bitpacked_without_patches = match bitpacked.bit_widths() { + BitWidths::Global(bw) => { + BitPacked::try_new(packed, ptype, validity, None, bw, len, offset)? + } + BitWidths::Blocked(block_offsets) => BitPacked::try_new_with_block_offsets( + packed, + ptype, + validity, + None, + block_offsets, + len, + offset, + )?, + } + .into_array(); + + let patched = Patched::from_array_and_patches( + bitpacked_without_patches, + &patches, + &mut session.create_execution_ctx(), + )?; + + Ok(patched.into_array()) + } + + fn is_supported_encoding(&self, id: &ArrayId) -> bool { + id == ArrayVTable::id(&BitPacked) || id == ArrayVTable::id(&Patched) + } +} diff --git a/encodings/fastlanes/src/bitpacking/plugin/tests.rs b/encodings/fastlanes/src/bitpacking/plugin/tests.rs new file mode 100644 index 00000000000..bb06d2d2fc9 --- /dev/null +++ b/encodings/fastlanes/src/bitpacking/plugin/tests.rs @@ -0,0 +1,419 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +use std::sync::LazyLock; + +use prost::Message; +use rstest::rstest; +use vortex_array::ArrayContext; +use vortex_array::ArrayDeserialization; +use vortex_array::ArrayPlugin; +use vortex_array::ArrayRef; +use vortex_array::IntoArray; +use vortex_array::VortexSessionExecute; +use vortex_array::arrays::PatchedArray; +use vortex_array::arrays::PrimitiveArray; +use vortex_array::arrays::patched::PatchedArraySlotsExt; +use vortex_array::assert_arrays_eq; +use vortex_array::buffer::BufferHandle; +use vortex_array::builtins::ArrayBuiltins; +use vortex_array::dtype::DType; +use vortex_array::dtype::Nullability; +use vortex_array::dtype::PType; +use vortex_array::serde::SerializeOptions; +use vortex_array::serde::SerializedArray; +use vortex_array::session::ArraySessionExt; +use vortex_array::validity::Validity; +use vortex_buffer::Buffer; +use vortex_buffer::ByteBufferMut; +use vortex_error::VortexResult; +use vortex_error::vortex_bail; +use vortex_error::vortex_err; +use vortex_session::VortexSession; +use vortex_session::registry::ReadContext; + +use super::BitPackedPatchedPlugin; +use super::BitPackedPlugin; +use super::bitpacked_v1_id; +use super::bitpacked_v2_id; +use super::v1::BitPackedMetadata; +use super::v2::BitPackedV2Metadata; +use crate::BitPacked; +use crate::BitPackedArray; +use crate::BitPackedArrayExt; +use crate::BitPackedData; +use crate::BitWidths; +use crate::bitpack_compress::bitpack_encode_blocked; +use crate::bitpack_compress::bitpack_to_best_bit_widths; + +static SESSION: LazyLock = LazyLock::new(|| { + let session = vortex_array::array_session(); + session.arrays().register(BitPackedPatchedPlugin); + session +}); + +#[test] +fn test_decode_bitpacked_patches() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + // Create values where some exceed the bit width, causing patches. + // With bit_width=9, max value is 511. Values >=512 become patches. + let values: Buffer = (0i32..=512).collect(); + let parray = values.into_array(); + let bitpacked = BitPackedData::encode(&parray, 9, &mut ctx)?; + + assert!( + bitpacked.patches().is_some(), + "Expected BitPacked array to have patches" + ); + + let array = bitpacked.as_array(); + + let serialization = SESSION.array_serialize(array)?.unwrap(); + let children = array.children(); + let buffers = array + .buffers() + .into_iter() + .map(BufferHandle::new_host) + .collect::>(); + + let deserialized = BitPackedPatchedPlugin.deserialize( + ArrayDeserialization::new( + bitpacked_v1_id(), + array.dtype(), + array.len(), + &serialization.metadata, + &buffers, + &children, + ), + &SESSION, + )?; + + let patched: PatchedArray = deserialized + .try_downcast() + .map_err(|a| vortex_err!("Expected Patched, got {}", a.encoding_id()))?; + + let inner_bitpacked: BitPackedArray = patched + .inner() + .clone() + .try_downcast() + .map_err(|a| vortex_err!("Expected inner BitPacked, got {}", a.encoding_id()))?; + + assert!( + inner_bitpacked.patches().is_none(), + "Inner BitPacked should NOT have patches" + ); + + Ok(()) +} + +#[test] +fn bitpacked_without_patches_stays_bitpacked() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + // With bit_width=16, max value is 65535. All values 0..100 fit. + let values: Buffer = (0i32..100).collect(); + let parray = values.into_array(); + let bitpacked = BitPackedData::encode(&parray, 16, &mut ctx)?; + + assert!( + bitpacked.patches().is_none(), + "Expected BitPacked array without patches" + ); + + let array = bitpacked.as_array(); + + let serialization = SESSION.array_serialize(array)?.unwrap(); + let children = array.children(); + let buffers = array + .buffers() + .into_iter() + .map(BufferHandle::new_host) + .collect::>(); + + let deserialized = BitPackedPatchedPlugin.deserialize( + ArrayDeserialization::new( + bitpacked_v1_id(), + array.dtype(), + array.len(), + &serialization.metadata, + &buffers, + &children, + ), + &SESSION, + )?; + + let result = deserialized + .try_downcast::() + .map_err(|a| vortex_err!("Expected deserialize BitPacked, got {}", a.encoding_id()))?; + + assert!(result.patches().is_none(), "Result should not have patches"); + + Ok(()) +} + +#[test] +fn primitive_array_returns_error() -> VortexResult<()> { + let array = PrimitiveArray::from_iter([1i32, 2, 3]).into_array(); + + let serialization = SESSION.array_serialize(&array)?.unwrap(); + let children = array.children(); + let buffers = array + .buffers() + .into_iter() + .map(BufferHandle::new_host) + .collect::>(); + + let result = BitPackedPatchedPlugin.deserialize( + ArrayDeserialization::new( + bitpacked_v1_id(), + array.dtype(), + array.len(), + &serialization.metadata, + &buffers, + &children, + ), + &SESSION, + ); + + assert!( + result.is_err(), + "Expected error when deserializing PrimitiveArray with BitPackedPatchedPlugin" + ); + + Ok(()) +} + +static PLUGIN_SESSION: LazyLock = LazyLock::new(|| { + let session = vortex_array::array_session(); + session.arrays().register(BitPackedPlugin); + session +}); + +fn roundtrip(array: &ArrayRef) -> VortexResult { + let array_ctx = ArrayContext::empty(); + let buffers = array.serialize(&array_ctx, &PLUGIN_SESSION, &SerializeOptions::default())?; + let mut bytes = ByteBufferMut::empty(); + for buffer in buffers { + bytes.extend_from_slice(&buffer); + } + SerializedArray::try_from(bytes.freeze())?.decode( + array.dtype(), + array.len(), + &ReadContext::new(array_ctx.to_ids()), + &PLUGIN_SESSION, + ) +} + +#[rstest] +#[case::no_patches(PrimitiveArray::from_iter(0u32..3000).into_array(), 12, 0..3000)] +#[case::patches(PrimitiveArray::from_iter(0i32..=2048).into_array(), 9, 0..2049)] +#[case::nullable( + PrimitiveArray::from_option_iter((0u16..3000).map(|i| (i % 5 != 0).then_some(i))).into_array(), + 8, + 0..3000, +)] +#[case::sliced(PrimitiveArray::from_iter(0u32..3000).into_array(), 12, 700..1900)] +fn serde_roundtrip( + #[case] values: ArrayRef, + #[case] bit_width: u8, + #[case] range: std::ops::Range, +) -> VortexResult<()> { + let mut ctx = PLUGIN_SESSION.create_execution_ctx(); + let array = BitPackedData::encode(&values, bit_width, &mut ctx)? + .into_array() + .slice(range.clone())?; + let view = array.as_::(); + let serialization = PLUGIN_SESSION + .array_serialize(&array)? + .ok_or_else(|| vortex_err!("BitPacked must serialize"))?; + assert_eq!(serialization.serialized_id, bitpacked_v1_id()); + let metadata = BitPackedMetadata::decode(serialization.metadata.as_slice())?; + assert_eq!(metadata.bit_width, u32::from(bit_width)); + assert_eq!(metadata.offset, u32::from(view.offset())); + assert_eq!(metadata.patches.is_some(), view.patches().is_some()); + + let read = roundtrip(&array)?; + assert_eq!(read.encoding_id(), BitPackedPlugin.id()); + assert_arrays_eq!(read, values.slice(range)?, &mut ctx); + Ok(()) +} + +#[test] +fn vtable_serde_requires_plugin() -> VortexResult<()> { + let values = PrimitiveArray::from_iter([1u8, 2, 3]).into_array(); + let array = BitPackedData::encode(&values, 2, &mut SESSION.create_execution_ctx())?; + let session = vortex_array::array_session(); + session.arrays().register(BitPacked); + assert!(session.array_serialize(&array.into_array()).is_err()); + Ok(()) +} + +/// Values whose 1024-value blocks need 2 to 6 bits. +fn drifting() -> PrimitiveArray { + PrimitiveArray::from_iter((0..5000u32).map(|i| i % (4 << (i / 1024)))) +} + +#[rstest] +#[case::no_patches(drifting(), None)] +#[case::patches(drifting(), Some(vec![2, 3, 4, 5, 3]))] +#[case::nullable( + PrimitiveArray::from_option_iter( + (0..5000u32).map(|i| (i % 7 != 0).then_some(i % (4 << (i / 1024)))), + ), + None, +)] +fn v2_roundtrip( + #[case] values: PrimitiveArray, + #[case] bit_widths: Option>, +) -> VortexResult<()> { + let mut ctx = PLUGIN_SESSION.create_execution_ctx(); + let array = match bit_widths { + Some(bit_widths) => bitpack_encode_blocked(&values, &bit_widths, None, &mut ctx)?, + None => bitpack_to_best_bit_widths(&values, &mut ctx)?, + }; + let BitWidths::Blocked(block_offsets) = array.bit_widths() else { + vortex_bail!("expected block offsets"); + }; + let serialization = PLUGIN_SESSION + .array_serialize(array.as_array())? + .ok_or_else(|| vortex_err!("BitPacked must serialize"))?; + assert_eq!(serialization.serialized_id, bitpacked_v2_id()); + let metadata = BitPackedV2Metadata::decode(serialization.metadata.as_slice())?; + assert_eq!(metadata.offset, 0); + assert_eq!( + metadata.block_offsets_ptype, + PType::try_from(block_offsets.dtype())? as i32 + ); + assert_eq!(metadata.patches.is_some(), array.patches().is_some()); + + let read = roundtrip(array.as_array())?; + assert_eq!(read.encoding_id(), bitpacked_v2_id()); + assert!(!read.as_::().bit_widths().is_global()); + assert_arrays_eq!(read, values, &mut ctx); + Ok(()) +} + +#[test] +fn v2_roundtrip_offset_and_nonzero_base() -> VortexResult<()> { + let mut ctx = PLUGIN_SESSION.create_execution_ctx(); + let values = drifting(); + let encoded = bitpack_to_best_bit_widths(&values, &mut ctx)?; + let BitWidths::Blocked(block_offsets) = encoded.bit_widths() else { + vortex_bail!("expected block offsets"); + }; + let block_offsets = block_offsets + .cast(DType::Primitive(PType::U64, Nullability::NonNullable))? + .execute::(&mut ctx)?; + let block_offsets = + PrimitiveArray::from_iter(block_offsets.as_slice::().iter().map(|o| o + 128)); + let array = BitPacked::try_new_with_block_offsets( + encoded.packed().clone(), + PType::U32, + Validity::NonNullable, + None, + block_offsets.into_array(), + 4900, + 17, + )? + .into_array(); + + let serialization = PLUGIN_SESSION + .array_serialize(&array)? + .ok_or_else(|| vortex_err!("BitPacked must serialize"))?; + assert_eq!( + BitPackedV2Metadata::decode(serialization.metadata.as_slice())?.offset, + 17 + ); + assert_arrays_eq!( + roundtrip(&array)?, + values.into_array().slice(17..4917)?, + &mut ctx + ); + Ok(()) +} + +#[test] +fn v2_rejects_malformed_parts() -> VortexResult<()> { + let array = + bitpack_to_best_bit_widths(&drifting(), &mut PLUGIN_SESSION.create_execution_ctx())?; + let BitWidths::Blocked(block_offsets) = array.bit_widths() else { + vortex_bail!("expected block offsets"); + }; + let metadata = BitPackedV2Metadata { + offset: 0, + block_offsets_ptype: PType::try_from(block_offsets.dtype())? as i32, + patches: None, + }; + let buffers = [array.packed().clone()]; + let deserialize = |metadata: &BitPackedV2Metadata, children: &[ArrayRef]| { + BitPackedPlugin.deserialize( + ArrayDeserialization::new( + bitpacked_v2_id(), + array.dtype(), + array.len(), + &metadata.encode_to_vec(), + &buffers, + &children, + ), + &PLUGIN_SESSION, + ) + }; + + assert!(deserialize(&metadata, std::slice::from_ref(&block_offsets)).is_ok()); + assert!(deserialize(&metadata, &[]).is_err()); + let offset_past_block = BitPackedV2Metadata { + offset: 1024, + block_offsets_ptype: metadata.block_offsets_ptype, + patches: None, + }; + assert!(deserialize(&offset_past_block, std::slice::from_ref(&block_offsets)).is_err()); + let signed = BitPackedV2Metadata { + offset: 0, + block_offsets_ptype: PType::I32 as i32, + patches: None, + }; + let signed_offsets = + block_offsets.cast(DType::Primitive(PType::I32, Nullability::NonNullable))?; + assert!(deserialize(&signed, &[signed_offsets]).is_err()); + Ok(()) +} + +#[test] +fn test_decode_blocked_bitpacked_patches() -> VortexResult<()> { + let mut ctx = SESSION.create_execution_ctx(); + let values = drifting(); + let array = bitpack_encode_blocked(&values, &[2, 3, 4, 5, 3], None, &mut ctx)?.into_array(); + + let serialization = SESSION.array_serialize(&array)?.unwrap(); + assert_eq!(serialization.serialized_id, bitpacked_v2_id()); + let children = array.children(); + let buffers = array + .buffers() + .into_iter() + .map(BufferHandle::new_host) + .collect::>(); + + let deserialized = BitPackedPatchedPlugin.deserialize( + ArrayDeserialization::new( + bitpacked_v2_id(), + array.dtype(), + array.len(), + &serialization.metadata, + &buffers, + &children, + ), + &SESSION, + )?; + + let patched: PatchedArray = deserialized + .try_downcast() + .map_err(|a| vortex_err!("Expected Patched, got {}", a.encoding_id()))?; + let inner_bitpacked: BitPackedArray = patched + .inner() + .clone() + .try_downcast() + .map_err(|a| vortex_err!("Expected inner BitPacked, got {}", a.encoding_id()))?; + assert!(inner_bitpacked.patches().is_none()); + assert!(!inner_bitpacked.bit_widths().is_global()); + assert_arrays_eq!(patched, values, &mut ctx); + Ok(()) +} diff --git a/encodings/fastlanes/src/bitpacking/plugin/v1.rs b/encodings/fastlanes/src/bitpacking/plugin/v1.rs new file mode 100644 index 00000000000..00ab87481ae --- /dev/null +++ b/encodings/fastlanes/src/bitpacking/plugin/v1.rs @@ -0,0 +1,142 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +//! Serde for `fastlanes.bitpacked`, which stores a single bit width in its metadata. + +use prost::Message; +use vortex_array::Array; +use vortex_array::ArrayDeserialization; +use vortex_array::ArrayParts; +use vortex_array::ArrayRef; +use vortex_array::ArraySerialization; +use vortex_array::ArraySlots; +use vortex_array::ArrayView; +use vortex_array::IntoArray; +use vortex_array::patches::Patches; +use vortex_array::patches::PatchesData; +use vortex_array::patches::PatchesMetadata; +use vortex_array::validity::Validity; +use vortex_array::vtable::validity_to_child; +use vortex_error::VortexResult; +use vortex_error::vortex_bail; +use vortex_error::vortex_err; + +use super::bitpacked_v1_id; +use crate::BitPacked; +use crate::BitPackedArrayExt; +use crate::BitPackedData; +use crate::bitpacking::array::BitPackedSlots; + +#[derive(Clone, prost::Message)] +pub struct BitPackedMetadata { + #[prost(uint32, tag = "1")] + pub(crate) bit_width: u32, + #[prost(uint32, tag = "2")] + pub(crate) offset: u32, // must be <1024 + #[prost(message, optional, tag = "3")] + pub(crate) patches: Option, +} + +pub(super) fn serialize( + array: &ArrayRef, + view: ArrayView<'_, BitPacked>, + bit_width: u8, +) -> VortexResult { + let metadata = BitPackedMetadata { + bit_width: u32::from(bit_width), + offset: view.offset() as u32, + patches: view + .patches() + .map(|p| p.to_metadata(view.len(), view.dtype())) + .transpose()?, + } + .encode_to_vec(); + Ok(ArraySerialization::from_array( + bitpacked_v1_id(), + array, + metadata, + )) +} + +pub(super) fn deserialize(parts: ArrayDeserialization<'_>) -> VortexResult { + let ArrayDeserialization { + dtype, + len, + metadata, + buffers, + children, + .. + } = parts; + + let metadata = BitPackedMetadata::decode(metadata)?; + if buffers.len() != 1 { + vortex_bail!("Expected 1 buffer, got {}", buffers.len()); + } + let packed = buffers[0].clone(); + + let load_validity = |child_idx: usize| { + if children.len() == child_idx { + Ok(Validity::from(dtype.nullability())) + } else if children.len() == child_idx + 1 { + let validity = children.get(child_idx, &Validity::DTYPE, len)?; + Ok(Validity::Array(validity)) + } else { + vortex_bail!( + "Expected {} or {} children, got {}", + child_idx, + child_idx + 1, + children.len() + ); + } + }; + + let validity_idx = match &metadata.patches { + None => 0, + Some(patches_meta) if patches_meta.chunk_offsets_dtype()?.is_some() => 3, + Some(_) => 2, + }; + + let validity = load_validity(validity_idx)?; + + let patches = metadata + .patches + .map(|p| { + let indices = children.get(0, &p.indices_dtype()?, p.len()?)?; + let values = children.get(1, dtype, p.len()?)?; + let chunk_offsets = p + .chunk_offsets_dtype()? + .map(|dtype| children.get(2, &dtype, p.chunk_offsets_len() as usize)) + .transpose()?; + + Patches::new(len, p.offset()?, indices, values, chunk_offsets) + }) + .transpose()?; + + let slots = { + let mut s = ArraySlots::with_capacity(BitPackedSlots::COUNT); + PatchesData::push_slots(&mut s, patches.as_ref()); + s.push(validity_to_child(&validity, len)); + s.push(None); + s + }; + let data = BitPackedData::try_new( + packed, + patches, + u8::try_from(metadata.bit_width).map_err(|_| { + vortex_err!( + "BitPackedMetadata bit_width {} does not fit in u8", + metadata.bit_width + ) + })?, + u16::try_from(metadata.offset).map_err(|_| { + vortex_err!( + "BitPackedMetadata offset {} does not fit in u16", + metadata.offset + ) + })?, + )?; + Ok(Array::::try_from_parts( + ArrayParts::new(BitPacked, dtype.clone(), len, data).with_slots(slots), + )? + .into_array()) +} diff --git a/encodings/fastlanes/src/bitpacking/plugin/v2.rs b/encodings/fastlanes/src/bitpacking/plugin/v2.rs new file mode 100644 index 00000000000..b8b6480e479 --- /dev/null +++ b/encodings/fastlanes/src/bitpacking/plugin/v2.rs @@ -0,0 +1,146 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +//! Serde for `fastlanes.bitpacked.v2`, which stores the byte boundaries of the packed blocks as a +//! child, so each block has its own bit width. + +use prost::Message; +use vortex_array::Array; +use vortex_array::ArrayDeserialization; +use vortex_array::ArrayParts; +use vortex_array::ArrayRef; +use vortex_array::ArraySerialization; +use vortex_array::ArraySlots; +use vortex_array::ArrayView; +use vortex_array::IntoArray; +use vortex_array::dtype::DType; +use vortex_array::dtype::Nullability; +use vortex_array::dtype::PType; +use vortex_array::patches::Patches; +use vortex_array::patches::PatchesData; +use vortex_array::patches::PatchesMetadata; +use vortex_array::validity::Validity; +use vortex_array::vtable::validity_to_child; +use vortex_error::VortexResult; +use vortex_error::vortex_bail; +use vortex_error::vortex_ensure; + +use super::bitpacked_v2_id; +use crate::BitPacked; +use crate::BitPackedArrayExt; +use crate::BitPackedData; +use crate::FL_CHUNK_SIZE; +use crate::bitpacking::array::BitPackedSlots; + +/// Metadata for `fastlanes.bitpacked.v2`. +/// +/// The children are the patch indices, values and chunk offsets when there are patches, then the +/// validity when there is a validity child, then the block offsets. +#[derive(Clone, prost::Message)] +pub(super) struct BitPackedV2Metadata { + /// The position of the first element within the first block. + #[prost(uint32, tag = "1")] + pub(super) offset: u32, + /// The unsigned integer type of the block offsets child. + #[prost(enumeration = "PType", tag = "2")] + pub(super) block_offsets_ptype: i32, + #[prost(message, optional, tag = "3")] + pub(super) patches: Option, +} + +pub(super) fn serialize( + array: &ArrayRef, + view: ArrayView<'_, BitPacked>, + block_offsets: &ArrayRef, +) -> VortexResult { + let metadata = BitPackedV2Metadata { + offset: u32::from(view.offset()), + block_offsets_ptype: PType::try_from(block_offsets.dtype())? as i32, + patches: view + .patches() + .map(|p| p.to_metadata(view.len(), view.dtype())) + .transpose()?, + }; + Ok(ArraySerialization::from_array( + bitpacked_v2_id(), + array, + metadata.encode_to_vec(), + )) +} + +pub(super) fn deserialize(parts: ArrayDeserialization<'_>) -> VortexResult { + let ArrayDeserialization { + dtype, + len, + metadata, + buffers, + children, + .. + } = parts; + + let metadata = BitPackedV2Metadata::decode(metadata)?; + vortex_ensure!( + buffers.len() == 1, + "Expected 1 buffer, got {}", + buffers.len() + ); + vortex_ensure!( + usize::try_from(metadata.offset).is_ok_and(|offset| offset < FL_CHUNK_SIZE), + "BitPacked offset must be less than {FL_CHUNK_SIZE}, got {}", + metadata.offset + ); + let offset = u16::try_from(metadata.offset)?; + let block_offsets_ptype = PType::try_from(metadata.block_offsets_ptype)?; + + let num_patch_children = match &metadata.patches { + None => 0, + Some(patches_meta) if patches_meta.chunk_offsets_dtype()?.is_some() => 3, + Some(_) => 2, + }; + let validity = if children.len() == num_patch_children + 1 { + Validity::from(dtype.nullability()) + } else if children.len() == num_patch_children + 2 { + Validity::Array(children.get(num_patch_children, &Validity::DTYPE, len)?) + } else { + vortex_bail!( + "Expected {} or {} children, got {}", + num_patch_children + 1, + num_patch_children + 2, + children.len() + ); + }; + + let num_blocks = (usize::from(offset) + len).div_ceil(FL_CHUNK_SIZE); + let block_offsets = children.get( + children.len() - 1, + &DType::Primitive(block_offsets_ptype, Nullability::NonNullable), + num_blocks + 1, + )?; + + let patches = metadata + .patches + .map(|p| { + let indices = children.get(0, &p.indices_dtype()?, p.len()?)?; + let values = children.get(1, dtype, p.len()?)?; + let chunk_offsets = p + .chunk_offsets_dtype()? + .map(|dtype| children.get(2, &dtype, p.chunk_offsets_len() as usize)) + .transpose()?; + + Patches::new(len, p.offset()?, indices, values, chunk_offsets) + }) + .transpose()?; + + let slots = { + let mut s = ArraySlots::with_capacity(BitPackedSlots::COUNT); + PatchesData::push_slots(&mut s, patches.as_ref()); + s.push(validity_to_child(&validity, len)); + s.push(Some(block_offsets)); + s + }; + let data = BitPackedData::try_new_blocked(buffers[0].clone(), patches, offset)?; + Ok(Array::::try_from_parts( + ArrayParts::new(BitPacked, dtype.clone(), len, data).with_slots(slots), + )? + .into_array()) +} diff --git a/encodings/fastlanes/src/bitpacking/vtable/mod.rs b/encodings/fastlanes/src/bitpacking/vtable/mod.rs index aa18bd289c6..522c0a803ec 100644 --- a/encodings/fastlanes/src/bitpacking/vtable/mod.rs +++ b/encodings/fastlanes/src/bitpacking/vtable/mod.rs @@ -36,16 +36,21 @@ use vortex_error::vortex_bail; use vortex_error::vortex_ensure; use vortex_error::vortex_panic; use vortex_session::VortexSession; -use vortex_session::registry::CachedId; use crate::BitPackedArrayExt; use crate::BitPackedData; use crate::BitPackedDataParts; +use crate::BitWidths; +use crate::FL_CHUNK_SIZE; use crate::bitpack_decompress::unpack_array; +use crate::bitpack_decompress::unpack_array_blocked; use crate::bitpack_decompress::unpack_into_primitive_builder; +use crate::bitpack_decompress::unpack_into_primitive_builder_blocked; +use crate::bitpacked_v2_id; use crate::bitpacking::array::BitPackedSlots; use crate::bitpacking::array::BitPackedSlotsView; use crate::bitpacking::array::PATCH_SLOTS; +use crate::bitpacking::array::validate_block_offsets; use crate::bitpacking::vtable::rules::RULES; mod kernels; mod operations; @@ -62,7 +67,7 @@ pub(crate) fn initialize(session: &VortexSession) { impl ArrayHash for BitPackedData { fn array_hash(&self, state: &mut H, accuracy: EqMode) { self.offset.hash(state); - self.bit_width.hash(state); + self.global_bit_width.hash(state); self.packed.array_hash(state, accuracy); self.patches_data.hash(state); } @@ -71,7 +76,7 @@ impl ArrayHash for BitPackedData { impl ArrayEq for BitPackedData { fn array_eq(&self, other: &Self, accuracy: EqMode) -> bool { self.offset == other.offset - && self.bit_width == other.bit_width + && self.global_bit_width == other.global_bit_width && self.packed.array_eq(&other.packed, accuracy) && self.patches_data == other.patches_data } @@ -84,8 +89,7 @@ impl VTable for BitPacked { type ValidityVTable = Self; fn id(&self) -> ArrayId { - static ID: CachedId = CachedId::new("fastlanes.bitpacked"); - *ID + bitpacked_v2_id() } fn validate( @@ -95,7 +99,25 @@ impl VTable for BitPacked { len: usize, slots: &[Option], ) -> VortexResult<()> { + vortex_ensure!( + slots.len() == BitPackedSlots::COUNT, + "Expected {} slots, got {}", + BitPackedSlots::COUNT, + slots.len() + ); let bp_slots = BitPackedSlotsView::from_slots(slots); + match (data.global_bit_width, bp_slots.block_offsets) { + (Some(_), None) => {} + (None, Some(block_offsets)) => validate_block_offsets( + block_offsets, + dtype.as_ptype(), + (len + data.offset as usize).div_ceil(FL_CHUNK_SIZE), + data.packed.len(), + )?, + _ => { + vortex_bail!("BitPacked needs exactly one of a global bit width and block offsets") + } + } let validity = child_to_validity(bp_slots.validity_child, dtype.nullability()); let patches = @@ -105,7 +127,7 @@ impl VTable for BitPacked { dtype.as_ptype(), &validity, patches.as_ref(), - data.bit_width, + data.global_bit_width, len, data.offset, ) @@ -171,15 +193,18 @@ impl VTable for BitPacked { builder: &mut dyn ArrayBuilder, ctx: &mut ExecutionCtx, ) -> VortexResult<()> { + let bit_widths = array.bit_widths(); match_each_integer_ptype!(array.dtype().as_ptype(), |T| { - unpack_into_primitive_builder::( - array, - builder - .as_any_mut() - .downcast_mut() - .vortex_expect("bit packed array must canonicalize into a primitive array"), - ctx, - ) + let builder = builder + .as_any_mut() + .downcast_mut() + .vortex_expect("bit packed array must canonicalize into a primitive array"); + match &bit_widths { + BitWidths::Global(_) => unpack_into_primitive_builder::(array, builder, ctx), + BitWidths::Blocked(offsets) => { + unpack_into_primitive_builder_blocked::(array, offsets, builder, ctx) + } + } }) } @@ -196,9 +221,11 @@ impl VTable for BitPacked { ); require_validity!(array, BitPackedSlots::VALIDITY_CHILD); - Ok(ExecutionResult::done( - unpack_array(array.as_view(), ctx)?.into_array(), - )) + let decoded = match array.bit_widths() { + BitWidths::Global(_) => unpack_array(array.as_view(), ctx)?, + BitWidths::Blocked(offsets) => unpack_array_blocked(array.as_view(), &offsets, ctx)?, + }; + Ok(ExecutionResult::done(decoded.into_array())) } fn reduce_parent( @@ -214,6 +241,7 @@ impl VTable for BitPacked { pub struct BitPacked; impl BitPacked { + /// Construct a bit-packed array whose blocks all use `bit_width`. pub fn try_new( packed: BufferHandle, ptype: PType, @@ -225,23 +253,52 @@ impl BitPacked { ) -> VortexResult { let dtype = DType::Primitive(ptype, validity.nullability()); let slots = { - let mut s = ArraySlots::with_capacity(4); + let mut s = ArraySlots::with_capacity(BitPackedSlots::COUNT); PatchesData::push_slots(&mut s, patches.as_ref()); s.push(validity_to_child(&validity, len)); + s.push(None); s }; let data = BitPackedData::try_new(packed, patches, bit_width, offset)?; Array::try_from_parts(ArrayParts::new(BitPacked, dtype, len, data).with_slots(slots)) } + /// Construct a bit-packed array from packed data and explicit block byte boundaries. + /// + /// `block_offsets` must be non-nullable unsigned integers with one boundary per block and a + /// trailing end boundary. Each block's bit width is derived from the distance between its + /// boundaries. + pub fn try_new_with_block_offsets( + packed: BufferHandle, + ptype: PType, + validity: Validity, + patches: Option, + block_offsets: ArrayRef, + len: usize, + offset: u16, + ) -> VortexResult { + let dtype = DType::Primitive(ptype, validity.nullability()); + let slots = { + let mut s = ArraySlots::with_capacity(BitPackedSlots::COUNT); + PatchesData::push_slots(&mut s, patches.as_ref()); + s.push(validity_to_child(&validity, len)); + s.push(Some(block_offsets)); + s + }; + let data = BitPackedData::try_new_blocked(packed, patches, offset)?; + Array::try_from_parts(ArrayParts::new(BitPacked, dtype, len, data).with_slots(slots)) + } + + /// Split the array into its parts. pub fn into_parts(array: BitPackedArray) -> BitPackedDataParts { let len = array.len(); let patches = array.patches(); let validity = array.validity().vortex_expect("BitPacked validity"); + let bit_widths = array.bit_widths(); let data = array.into_data(); BitPackedDataParts { offset: data.offset, - bit_width: data.bit_width, + bit_widths, len, packed: data.packed, patches, diff --git a/encodings/fastlanes/src/bitpacking/vtable/operations.rs b/encodings/fastlanes/src/bitpacking/vtable/operations.rs index 2816407ac03..428f9089499 100644 --- a/encodings/fastlanes/src/bitpacking/vtable/operations.rs +++ b/encodings/fastlanes/src/bitpacking/vtable/operations.rs @@ -8,6 +8,7 @@ use vortex_array::vtable::OperationsVTable; use vortex_error::VortexResult; use crate::BitPacked; +use crate::BitWidths; use crate::bitpack_decompress; use crate::bitpacking::array::BitPackedArrayExt; impl OperationsVTable for BitPacked { @@ -16,7 +17,7 @@ impl OperationsVTable for BitPacked { fn scalar_at( array: ArrayView<'_, BitPacked>, index: usize, - _ctx: &mut ExecutionCtx, + ctx: &mut ExecutionCtx, ) -> VortexResult { Ok( if let Some(patches) = array.patches() @@ -24,7 +25,12 @@ impl OperationsVTable for BitPacked { { patch } else { - bitpack_decompress::unpack_single(array, index) + match array.bit_widths() { + BitWidths::Global(_) => bitpack_decompress::unpack_single(array, index)?, + BitWidths::Blocked(offsets) => { + bitpack_decompress::unpack_single_blocked(array, &offsets, index, ctx)? + } + } }, ) } diff --git a/encodings/fastlanes/src/for_/array/for_decompress.rs b/encodings/fastlanes/src/for_/array/for_decompress.rs index b07d4d48df0..91ab330e27b 100644 --- a/encodings/fastlanes/src/for_/array/for_decompress.rs +++ b/encodings/fastlanes/src/for_/array/for_decompress.rs @@ -28,10 +28,12 @@ use vortex_buffer::BufferAllocatorRef; use vortex_buffer::BufferMut; use vortex_error::VortexExpect; use vortex_error::VortexResult; +use vortex_error::vortex_bail; use vortex_error::vortex_err; use crate::BitPacked; use crate::BitPackedArrayExt; +use crate::BitWidths; use crate::FL_CHUNK_SIZE; use crate::FoRArray; use crate::for_::array::FoRArrayExt; @@ -283,7 +285,10 @@ fn unpack_chunks< output: &mut [MaybeUninit], ) -> VortexResult<()> { let offset = usize::from(bp.offset()); - let bit_width = bp.bit_width() as usize; + let BitWidths::Global(bit_width) = bp.bit_widths() else { + vortex_bail!("BitPacked array has per-block bit widths"); + }; + let bit_width = bit_width as usize; // SAFETY: `T::Physical` is `T` with the same size and alignment, and the unpack is the same // wrapping addition in two's complement whichever signedness `T` has. let output = diff --git a/encodings/fastlanes/src/for_/vtable/mod.rs b/encodings/fastlanes/src/for_/vtable/mod.rs index 9a24362d7ed..9c85309f283 100644 --- a/encodings/fastlanes/src/for_/vtable/mod.rs +++ b/encodings/fastlanes/src/for_/vtable/mod.rs @@ -34,6 +34,7 @@ use vortex_error::vortex_panic; use vortex_session::VortexSession; use crate::BitPacked; +use crate::BitPackedArrayExt; use crate::FoRData; use crate::for_::array::FoRArrayExt; use crate::for_::array::FoRArraySlotsExt; @@ -148,9 +149,11 @@ impl VTable for FoR { require_child!(array, array.references(), FoRSlots::REFERENCES => Primitive) }; // The fused unpack reads a bit-packed child's buffers directly. Its chunks line up with - // the FoR chunks when the references are constant or the offsets match. + // the FoR chunks when the references are constant or the offsets match. It also needs a + // global bit width. let fused = array.encoded().as_opt::().is_some_and(|bp| { - array.constant_reference().is_some() || bp.offset() == array.offset() + bp.bit_widths().is_global() + && (array.constant_reference().is_some() || bp.offset() == array.offset()) }); let array = if fused { array diff --git a/vortex-btrblocks/src/builder.rs b/vortex-btrblocks/src/builder.rs index 7bdc6661ff6..d3dea3fdb53 100644 --- a/vortex-btrblocks/src/builder.rs +++ b/vortex-btrblocks/src/builder.rs @@ -5,6 +5,7 @@ use vortex_array::ArrayId; use vortex_decimal_byte_parts::decimal_byte_parts_v2_id; +use vortex_fastlanes::bitpacked_v2_id; use vortex_fastlanes::for_v2_id; use vortex_session::VortexSession; use vortex_utils::aliases::hash_set::HashSet; @@ -94,9 +95,9 @@ impl CompressionMode { fn excluded_encodings(self) -> Vec { match self { Self::All | Self::Default | Self::Compact => Vec::new(), - // Multi-part DecimalByteParts arrays and FoR arrays with per-chunk references have no - // CUDA decode kernel. - Self::Cuda => vec![decimal_byte_parts_v2_id(), for_v2_id()], + // Multi-part DecimalByteParts arrays, FoR arrays with per-chunk references and + // BitPacked arrays with per-block bit widths have no CUDA decode kernel. + Self::Cuda => vec![decimal_byte_parts_v2_id(), for_v2_id(), bitpacked_v2_id()], } } } diff --git a/vortex-btrblocks/src/schemes/integer/bitpacking.rs b/vortex-btrblocks/src/schemes/integer/bitpacking.rs index 5ac7d0e4078..adda3be4186 100644 --- a/vortex-btrblocks/src/schemes/integer/bitpacking.rs +++ b/vortex-btrblocks/src/schemes/integer/bitpacking.rs @@ -16,20 +16,69 @@ use vortex_compressor::scheme::CompressionEstimate; use vortex_compressor::scheme::DeferredEstimate; use vortex_compressor::scheme::EstimateVerdict; use vortex_error::VortexResult; +use vortex_error::vortex_bail; use vortex_fastlanes::BitPacked; +use vortex_fastlanes::BitWidths; use vortex_fastlanes::bitpack_compress::bit_width_histogram; use vortex_fastlanes::bitpack_compress::bitpack_encode; +use vortex_fastlanes::bitpack_compress::bitpack_to_best_bit_widths; use vortex_fastlanes::bitpack_compress::find_best_bit_width; +use vortex_fastlanes::bitpacked_v1_id; +use vortex_fastlanes::bitpacked_v2_id; use crate::ArrayAndStats; use crate::CascadingCompressor; use crate::CompressorContext; use crate::Scheme; +use crate::SchemeExt; use crate::compress_patches; +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +enum BitPackingSchemeMode { + V1, + V2, +} + +/// The BitPacking scheme that only produces arrays with one bit width. +pub(crate) static BITPACKING_V1: BitPackingScheme = BitPackingScheme::v1(); + +/// The BitPacking scheme that may give each 1024-element block its own bit width. +pub(crate) static BITPACKING_V2: BitPackingScheme = BitPackingScheme::v2(); + /// BitPacking encoding for non-negative integers. +/// +/// The v1 mode packs every value at one bit width, and serializes as `fastlanes.bitpacked`. The v2 +/// mode always chooses a width for each 1024-element block, and serializes as +/// `fastlanes.bitpacked.v2`. +/// +/// The default uses v1. [`refine`](Scheme::refine) picks v2 when the v2 ID is allowed and v1 +/// otherwise. #[derive(Debug, Copy, Clone, PartialEq, Eq)] -pub struct BitPackingScheme; +pub struct BitPackingScheme { + mode: BitPackingSchemeMode, +} + +impl BitPackingScheme { + /// Creates a BitPacking scheme configured for v1, which uses one bit width. + pub const fn v1() -> Self { + Self { + mode: BitPackingSchemeMode::V1, + } + } + + /// Creates a BitPacking scheme configured for v2, which may use one bit width per block. + pub const fn v2() -> Self { + Self { + mode: BitPackingSchemeMode::V2, + } + } +} + +impl Default for BitPackingScheme { + fn default() -> Self { + Self::v1() + } +} impl Scheme for BitPackingScheme { fn scheme_name(&self) -> &'static str { @@ -41,13 +90,33 @@ impl Scheme for BitPackingScheme { } fn produced_encodings(&self) -> Vec { - let mut encodings = vec![BitPacked.id()]; + let mut encodings = match self.mode { + // Global-width arrays serialize under the frozen v1 ID. + BitPackingSchemeMode::V1 => vec![bitpacked_v1_id()], + BitPackingSchemeMode::V2 => vec![bitpacked_v2_id()], + }; if use_experimental_patches() { encodings.push(Patched.id()); } encodings } + fn refine(&self, allowed: &dyn Fn(&ArrayId) -> bool) -> &dyn Scheme { + if allowed(&bitpacked_v2_id()) { + &BITPACKING_V2 + } else { + &BITPACKING_V1 + } + } + + /// Children: block offsets=0 in v2 mode. + fn num_children(&self) -> usize { + match self.mode { + BitPackingSchemeMode::V1 => 0, + BitPackingSchemeMode::V2 => 1, + } + } + fn expected_compression_ratio( &self, data: &ArrayAndStats, @@ -66,11 +135,15 @@ impl Scheme for BitPackingScheme { fn compress( &self, - _compressor: &CascadingCompressor, + compressor: &CascadingCompressor, data: &ArrayAndStats, - _compress_ctx: CompressorContext, + compress_ctx: CompressorContext, exec_ctx: &mut ExecutionCtx, ) -> VortexResult { + if self.mode == BitPackingSchemeMode::V2 { + return self.compress_blocked(compressor, data, compress_ctx, exec_ctx); + } + let primitive_array = data.array_as_primitive(); let histogram = bit_width_histogram(primitive_array, exec_ctx)?; @@ -97,7 +170,7 @@ impl Scheme for BitPackingScheme { ptype, parts.validity, None, - parts.bit_width, + bw, parts.len, parts.offset, )? @@ -122,7 +195,72 @@ impl Scheme for BitPackingScheme { ptype, parts.validity, parts.patches, - parts.bit_width, + bw, + parts.len, + parts.offset, + )? + .with_stats_set(packed_stats) + .into_array() + }; + + Ok(array) + } +} + +impl BitPackingScheme { + /// Bit-pack each 1024-element block at its own best width. + fn compress_blocked( + &self, + compressor: &CascadingCompressor, + data: &ArrayAndStats, + compress_ctx: CompressorContext, + exec_ctx: &mut ExecutionCtx, + ) -> VortexResult { + let primitive_array = data.array_as_primitive().into_owned(); + let packed = bitpack_to_best_bit_widths(&primitive_array, exec_ctx)?; + + let packed_stats = packed.statistics().to_owned(); + let ptype = packed.dtype().as_ptype(); + let mut parts = BitPacked::into_parts(packed); + let BitWidths::Blocked(block_offsets) = parts.bit_widths else { + vortex_bail!("Blocked bit-packing must produce block offsets"); + }; + let block_offsets = + compressor.compress_child(&block_offsets, &compress_ctx, self.id(), 0, exec_ctx)?; + + let array = if use_experimental_patches() { + let patches = parts.patches.take(); + // Transpose patches into G-ALP style PatchedArray, wrapping an inner BitPackedArray. + let array = BitPacked::try_new_with_block_offsets( + parts.packed, + ptype, + parts.validity, + None, + block_offsets, + parts.len, + parts.offset, + )? + .into_array(); + + match patches { + None => array, + Some(p) => Patched::from_array_and_patches(array, &p, exec_ctx)? + .with_stats_set(packed_stats) + .into_array(), + } + } else { + // Compress patches and place back into BitPackedArray. + let patches = parts + .patches + .take() + .map(|p| compress_patches(p, exec_ctx)) + .transpose()?; + BitPacked::try_new_with_block_offsets( + parts.packed, + ptype, + parts.validity, + patches, + block_offsets, parts.len, parts.offset, )? @@ -133,3 +271,41 @@ impl Scheme for BitPackingScheme { Ok(array) } } + +#[cfg(test)] +mod tests { + use rstest::rstest; + use vortex_array::ArrayId; + use vortex_fastlanes::bitpacked_v1_id; + use vortex_fastlanes::bitpacked_v2_id; + + use super::BITPACKING_V1; + use super::BITPACKING_V2; + use crate::Scheme; + use crate::SchemeExt; + + /// Both variants refine to v2 exactly when the v2 ID is allowed. + #[rstest] + #[case::neither(false, false, false)] + #[case::v1(true, false, false)] + #[case::v2_only(false, true, true)] + #[case::both(true, true, true)] + fn refine_picks_v2_when_v2_id_is_allowed( + #[case] allow_v1: bool, + #[case] allow_v2: bool, + #[case] expect_v2: bool, + #[values(&BITPACKING_V1, &BITPACKING_V2)] scheme: &'static dyn Scheme, + ) { + let allowed = |id: &ArrayId| { + (allow_v1 && *id == bitpacked_v1_id()) || (allow_v2 && *id == bitpacked_v2_id()) + }; + let refined = scheme.refine(&allowed); + assert_eq!(refined.id(), scheme.id()); + let expected: &dyn Scheme = if expect_v2 { + &BITPACKING_V2 + } else { + &BITPACKING_V1 + }; + assert_eq!(refined.produced_encodings(), expected.produced_encodings()); + } +} diff --git a/vortex-btrblocks/src/schemes/integer/for_.rs b/vortex-btrblocks/src/schemes/integer/for_.rs index 43c7ee56d68..34376673c11 100644 --- a/vortex-btrblocks/src/schemes/integer/for_.rs +++ b/vortex-btrblocks/src/schemes/integer/for_.rs @@ -25,7 +25,7 @@ use vortex_fastlanes::FoRArraySlotsExt; use vortex_fastlanes::for_v1_id; use vortex_fastlanes::for_v2_id; -use super::BitPackingScheme; +use super::BITPACKING_V1; use crate::ArrayAndStats; use crate::CascadingCompressor; use crate::CompressorContext; @@ -216,7 +216,7 @@ impl Scheme for FoRScheme { let leaf_ctx = compress_ctx.clone().as_leaf(); let biased_data = ArrayAndStats::new(biased.into_array(), compress_ctx.merged_stats_options()); - let compressed = BitPackingScheme.compress(compressor, &biased_data, leaf_ctx, exec_ctx)?; + let compressed = BITPACKING_V1.compress(compressor, &biased_data, leaf_ctx, exec_ctx)?; // TODO(connor): This should really be `new_unchecked`. let for_compressed = match for_array.constant_reference() { diff --git a/vortex-btrblocks/src/schemes/integer/mod.rs b/vortex-btrblocks/src/schemes/integer/mod.rs index 742b453c8ef..74edd53463f 100644 --- a/vortex-btrblocks/src/schemes/integer/mod.rs +++ b/vortex-btrblocks/src/schemes/integer/mod.rs @@ -15,6 +15,7 @@ mod zigzag; #[cfg(feature = "pco")] mod pco; +pub(crate) use bitpacking::BITPACKING_V1; pub use bitpacking::BitPackingScheme; pub use delta::DeltaScheme; pub(crate) use for_::FOR_V1; diff --git a/vortex-btrblocks/src/session.rs b/vortex-btrblocks/src/session.rs index 4ca2a0e5de3..120345efd62 100644 --- a/vortex-btrblocks/src/session.rs +++ b/vortex-btrblocks/src/session.rs @@ -48,7 +48,7 @@ impl Default for CompressionSession { &integer::FOR_V1, // NOTE: ZigZag should precede BitPacking because we don't want negative numbers. &integer::ZigZagScheme, - &integer::BitPackingScheme, + &integer::BITPACKING_V1, &integer::SparseScheme, &integer::IntDictScheme, &integer::RunEndScheme, diff --git a/vortex-btrblocks/tests/bitpacking_config.rs b/vortex-btrblocks/tests/bitpacking_config.rs new file mode 100644 index 00000000000..37621fed83c --- /dev/null +++ b/vortex-btrblocks/tests/bitpacking_config.rs @@ -0,0 +1,141 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +//! BitPacking scheme refinement by allowed serialized IDs, and per-block bit widths. + +#![cfg(test)] + +use std::sync::LazyLock; + +use rstest::rstest; +use vortex_array::ArrayContext; +use vortex_array::ArrayId; +use vortex_array::ArrayRef; +use vortex_array::IntoArray; +use vortex_array::VortexSessionExecute; +use vortex_array::arrays::PrimitiveArray; +use vortex_array::assert_arrays_eq; +use vortex_array::serde::SerializeOptions; +use vortex_array::serde::SerializedArray; +use vortex_btrblocks::BtrBlocksCompressorBuilder; +use vortex_btrblocks::schemes::integer::BitPackingScheme; +use vortex_buffer::ByteBufferMut; +use vortex_edition::EDITION_DECLARATIONS; +use vortex_edition::EDITION_FAMILIES; +use vortex_edition::EditionSession; +use vortex_edition::EditionSessionExt; +use vortex_edition::declarations::core::CORE_2026_08_3; +use vortex_error::VortexResult; +use vortex_fastlanes::bitpacked_v1_id; +use vortex_fastlanes::bitpacked_v2_id; +use vortex_session::VortexSession; +use vortex_session::registry::ReadContext; + +static BITPACKING_V1: BitPackingScheme = BitPackingScheme::v1(); + +/// Registers the fastlanes encodings, and enables no editions. +static SESSION: LazyLock = LazyLock::new(|| { + let session = vortex_array::array_session(); + vortex_fastlanes::initialize(&session); + session +}); + +/// Like [`SESSION`], with the latest core edition enabled: it allows `fastlanes.bitpacked` but not +/// `fastlanes.bitpacked.v2`. +static CORE_SESSION: LazyLock = LazyLock::new(|| { + let session = vortex_array::array_session().with::(); + for family in EDITION_FAMILIES { + session + .editions() + .declare_family(family) + .expect("first-party edition family"); + } + for declaration in EDITION_DECLARATIONS { + session + .register_edition(declaration) + .expect("first-party edition"); + } + session + .enable_edition(CORE_2026_08_3) + .expect("core edition is registered"); + vortex_fastlanes::initialize(&session); + session +}); + +/// Values whose 1024-value blocks need 1 to 8 bits. +fn drifting() -> ArrayRef { + PrimitiveArray::from_iter((0..8192u32).map(|i| i % (2 << (i / 1024)))).into_array() +} + +/// Values that need 7 bits in every block. +fn uniform() -> ArrayRef { + PrimitiveArray::from_iter((0..8192u32).map(|i| i % 128)).into_array() +} + +/// Compresses `array`, round trips it through serialization, and returns the serialized IDs. +fn compress_roundtrip( + builder: BtrBlocksCompressorBuilder, + array: &ArrayRef, +) -> VortexResult> { + let mut ctx = SESSION.create_execution_ctx(); + let compressed = builder.build().compress(array, &mut ctx)?; + assert_arrays_eq!(array, compressed, &mut ctx); + + let array_ctx = ArrayContext::empty(); + let mut bytes = ByteBufferMut::empty(); + for buffer in compressed.serialize(&array_ctx, &SESSION, &SerializeOptions::default())? { + bytes.extend_from_slice(&buffer); + } + let read = SerializedArray::try_from(bytes.freeze())?.decode( + array.dtype(), + array.len(), + &ReadContext::new(array_ctx.to_ids()), + &SESSION, + )?; + assert_arrays_eq!(array, read, &mut ctx); + Ok(array_ctx.to_ids()) +} + +/// Only BitPacking, with every serialized ID allowed, so it refines to v2. +fn bitpacking_only() -> BtrBlocksCompressorBuilder { + BtrBlocksCompressorBuilder::empty().with_new_scheme(&BITPACKING_V1) +} + +/// v2 produces per-block bit widths even when every block chooses the same width. +#[rstest] +#[case::drifting(drifting())] +#[case::uniform(uniform())] +fn v2_always_serializes_as_v2(#[case] array: ArrayRef) -> VortexResult<()> { + let ids = compress_roundtrip(bitpacking_only(), &array)?; + assert!(ids.contains(&bitpacked_v2_id())); + Ok(()) +} + +#[test] +fn nullable_drifting_roundtrip() -> VortexResult<()> { + let array = PrimitiveArray::from_option_iter( + (0..8192u64).map(|i| (i % 7 != 0).then_some(i % (2 << (i / 1024)))), + ) + .into_array(); + let ids = compress_roundtrip(bitpacking_only(), &array)?; + assert!(ids.contains(&bitpacked_v2_id())); + Ok(()) +} + +#[test] +fn core_edition_keeps_global_width() -> VortexResult<()> { + let ids = compress_roundtrip( + BtrBlocksCompressorBuilder::from_session(&CORE_SESSION), + &drifting(), + )?; + assert!(!ids.contains(&bitpacked_v2_id())); + Ok(()) +} + +#[test] +fn cuda_preset_keeps_global_width() -> VortexResult<()> { + let ids = compress_roundtrip(bitpacking_only().only_cuda_compatible(), &drifting())?; + assert!(ids.contains(&bitpacked_v1_id())); + assert!(!ids.contains(&bitpacked_v2_id())); + Ok(()) +} diff --git a/vortex-btrblocks/tests/decimal_config.rs b/vortex-btrblocks/tests/decimal_config.rs index e91995e0e4f..749aebc561b 100644 --- a/vortex-btrblocks/tests/decimal_config.rs +++ b/vortex-btrblocks/tests/decimal_config.rs @@ -45,6 +45,7 @@ use vortex_session::registry::ReadContext; static DECIMAL_V2: DecimalScheme = DecimalScheme::v2(); static FOR_V1: FoRScheme = FoRScheme::v1(); +static BITPACKING_V1: BitPackingScheme = BitPackingScheme::v1(); /// Registers the decimal and fastlanes encodings, and enables no editions. static SESSION: LazyLock = LazyLock::new(|| { @@ -199,7 +200,7 @@ fn wide_decimal_parts_roundtrip( if compress_children { builder = builder .with_new_scheme(&FOR_V1) - .with_new_scheme(&BitPackingScheme); + .with_new_scheme(&BITPACKING_V1); } let mut ctx = SESSION.create_execution_ctx(); let compressed = builder.build().compress(&array, &mut ctx)?; diff --git a/vortex-btrblocks/tests/for_config.rs b/vortex-btrblocks/tests/for_config.rs index 812e462871c..cf6be6e4826 100644 --- a/vortex-btrblocks/tests/for_config.rs +++ b/vortex-btrblocks/tests/for_config.rs @@ -33,7 +33,7 @@ use vortex_session::VortexSession; use vortex_session::registry::ReadContext; static FOR_V1: FoRScheme = FoRScheme::v1(); -static BITPACKING: BitPackingScheme = BitPackingScheme; +static BITPACKING: BitPackingScheme = BitPackingScheme::v1(); /// Registers the fastlanes encodings, and enables no editions. static SESSION: LazyLock = LazyLock::new(|| { diff --git a/vortex-btrblocks/tests/snapshots/golden__compact__binary_low_cardinality.snap b/vortex-btrblocks/tests/snapshots/golden__compact__binary_low_cardinality.snap index 788cf54ecd8..b53b1016145 100644 --- a/vortex-btrblocks/tests/snapshots/golden__compact__binary_low_cardinality.snap +++ b/vortex-btrblocks/tests/snapshots/golden__compact__binary_low_cardinality.snap @@ -5,7 +5,7 @@ expression: rendered input: binary, len=16384, nbytes=315856 root: vortex.dict(binary, len=16384) nbytes=6196 metadata: all_values_referenced: true - codes: fastlanes.bitpacked(u8, len=16384) nbytes=6144 + codes: fastlanes.bitpacked.v2(u8, len=16384) nbytes=6144 metadata: bit_width: 3, offset: 0 values: vortex.varbin(binary, len=5) nbytes=52 metadata: diff --git a/vortex-btrblocks/tests/snapshots/golden__compact__float_low_cardinality.snap b/vortex-btrblocks/tests/snapshots/golden__compact__float_low_cardinality.snap index 006f5a93b69..e6e54062b8e 100644 --- a/vortex-btrblocks/tests/snapshots/golden__compact__float_low_cardinality.snap +++ b/vortex-btrblocks/tests/snapshots/golden__compact__float_low_cardinality.snap @@ -5,7 +5,7 @@ expression: rendered input: f64, len=16384, nbytes=131072 root: vortex.dict(f64, len=16384) nbytes=6184 metadata: all_values_referenced: true - codes: fastlanes.bitpacked(u8, len=16384) nbytes=6144 + codes: fastlanes.bitpacked.v2(u8, len=16384) nbytes=6144 metadata: bit_width: 3, offset: 0 values: vortex.alp(f64, len=8) nbytes=40 metadata: exponents: e: 16, f: 11 diff --git a/vortex-btrblocks/tests/snapshots/golden__compact__int_low_cardinality.snap b/vortex-btrblocks/tests/snapshots/golden__compact__int_low_cardinality.snap index 0cbe2e06814..677fd20d6d1 100644 --- a/vortex-btrblocks/tests/snapshots/golden__compact__int_low_cardinality.snap +++ b/vortex-btrblocks/tests/snapshots/golden__compact__int_low_cardinality.snap @@ -5,7 +5,7 @@ expression: rendered input: i64, len=16384, nbytes=131072 root: vortex.dict(i64, len=16384) nbytes=6177 metadata: all_values_referenced: true - codes: fastlanes.bitpacked(u8, len=16384) nbytes=6144 + codes: fastlanes.bitpacked.v2(u8, len=16384) nbytes=6144 metadata: bit_width: 3, offset: 0 values: vortex.pco(i64, len=6) nbytes=33 metadata: ptype: i64, nrows: 6, slice: 0..6 diff --git a/vortex-btrblocks/tests/snapshots/golden__compact__int_negatives.snap b/vortex-btrblocks/tests/snapshots/golden__compact__int_negatives.snap index 71969bd5250..82c896b6082 100644 --- a/vortex-btrblocks/tests/snapshots/golden__compact__int_negatives.snap +++ b/vortex-btrblocks/tests/snapshots/golden__compact__int_negatives.snap @@ -5,7 +5,7 @@ expression: rendered input: i64, len=16384, nbytes=131072 root: fastlanes.for.v2(i64, len=16384) nbytes=16387 metadata: offset: 0 - encoded: fastlanes.bitpacked(i64, len=16384) nbytes=16384 + encoded: fastlanes.bitpacked.v2(i64, len=16384) nbytes=16384 metadata: bit_width: 8, offset: 0 references: vortex.constant(i64, len=16) nbytes=3 metadata: scalar: -128i64 diff --git a/vortex-btrblocks/tests/snapshots/golden__compact__string_low_cardinality.snap b/vortex-btrblocks/tests/snapshots/golden__compact__string_low_cardinality.snap index a65054ac178..1d5b1b7eab3 100644 --- a/vortex-btrblocks/tests/snapshots/golden__compact__string_low_cardinality.snap +++ b/vortex-btrblocks/tests/snapshots/golden__compact__string_low_cardinality.snap @@ -5,7 +5,7 @@ expression: rendered input: utf8, len=16384, nbytes=262144 root: vortex.dict(utf8, len=16384) nbytes=8287 metadata: all_values_referenced: true - codes: fastlanes.bitpacked(u8, len=16384) nbytes=8192 + codes: fastlanes.bitpacked.v2(u8, len=16384) nbytes=8192 metadata: bit_width: 4, offset: 0 values: vortex.zstd(utf8, len=12) nbytes=95 metadata: nrows: 12, slice: 0..12 diff --git a/vortex-btrblocks/tests/snapshots/golden__compact__struct_mixed.snap b/vortex-btrblocks/tests/snapshots/golden__compact__struct_mixed.snap index 83e34c583dc..ba975ae3c38 100644 --- a/vortex-btrblocks/tests/snapshots/golden__compact__struct_mixed.snap +++ b/vortex-btrblocks/tests/snapshots/golden__compact__struct_mixed.snap @@ -9,7 +9,7 @@ root: vortex.struct({id=i64, category=utf8, value=f64}, len=16384) nbytes=55986 metadata: base: 10000i64, multiplier: 7i64 category: vortex.dict(utf8, len=16384) nbytes=8287 metadata: all_values_referenced: true - codes: fastlanes.bitpacked(u8, len=16384) nbytes=8192 + codes: fastlanes.bitpacked.v2(u8, len=16384) nbytes=8192 metadata: bit_width: 4, offset: 0 values: vortex.zstd(utf8, len=12) nbytes=95 metadata: nrows: 12, slice: 0..12 diff --git a/vortex-btrblocks/tests/snapshots/golden__onpair__string_fsst_structured.snap b/vortex-btrblocks/tests/snapshots/golden__onpair__string_fsst_structured.snap index 028dbe81dc9..b20e50ae061 100644 --- a/vortex-btrblocks/tests/snapshots/golden__onpair__string_fsst_structured.snap +++ b/vortex-btrblocks/tests/snapshots/golden__onpair__string_fsst_structured.snap @@ -7,7 +7,7 @@ root: vortex.onpair(utf8, len=16384) nbytes=139746 metadata: dict_bytes_len: 11866 dict_offsets: vortex.primitive(u16, len=1626) nbytes=3252 metadata: ptype: u16 - codes: fastlanes.bitpacked(u16, len=63845) nbytes=88704 + codes: fastlanes.bitpacked.v2(u16, len=63845) nbytes=88704 metadata: bit_width: 11, offset: 0 codes_offsets: vortex.primitive(u16, len=16385) nbytes=32770 metadata: ptype: u16 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__binary_low_cardinality.snap b/vortex-btrblocks/tests/snapshots/golden__regular__binary_low_cardinality.snap index 788cf54ecd8..b53b1016145 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__binary_low_cardinality.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__binary_low_cardinality.snap @@ -5,7 +5,7 @@ expression: rendered input: binary, len=16384, nbytes=315856 root: vortex.dict(binary, len=16384) nbytes=6196 metadata: all_values_referenced: true - codes: fastlanes.bitpacked(u8, len=16384) nbytes=6144 + codes: fastlanes.bitpacked.v2(u8, len=16384) nbytes=6144 metadata: bit_width: 3, offset: 0 values: vortex.varbin(binary, len=5) nbytes=52 metadata: diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__decimal_prices.snap b/vortex-btrblocks/tests/snapshots/golden__regular__decimal_prices.snap index 6eb4ccfed8d..039178e7c35 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__decimal_prices.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__decimal_prices.snap @@ -5,5 +5,5 @@ expression: rendered input: decimal(12,2), len=16384, nbytes=131072 root: vortex.decimal_byte_parts.v2(decimal(12,2), len=16384) nbytes=49152 metadata: - msp: fastlanes.bitpacked(i32, len=16384) nbytes=49152 + msp: fastlanes.bitpacked.v2(i32, len=16384) nbytes=49152 metadata: bit_width: 24, offset: 0 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__float_alp_prices.snap b/vortex-btrblocks/tests/snapshots/golden__regular__float_alp_prices.snap index f70d90a301d..06364e4ddc6 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__float_alp_prices.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__float_alp_prices.snap @@ -5,5 +5,5 @@ expression: rendered input: f64, len=16384, nbytes=131072 root: vortex.alp(f64, len=16384) nbytes=49152 metadata: exponents: e: 14, f: 12 - encoded: fastlanes.bitpacked(i64, len=16384) nbytes=49152 + encoded: fastlanes.bitpacked.v2(i64, len=16384) nbytes=49152 metadata: bit_width: 24, offset: 0 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__float_full_precision.snap b/vortex-btrblocks/tests/snapshots/golden__regular__float_full_precision.snap index 487e7c7900b..319e63ef5fa 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__float_full_precision.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__float_full_precision.snap @@ -5,9 +5,9 @@ expression: rendered input: f64, len=16384, nbytes=131072 root: vortex.alprd(f64, len=16384) nbytes=112880 metadata: right_bit_width: 52, patch_offset: 0 - left_parts: fastlanes.bitpacked(u16, len=16384) nbytes=6144 + left_parts: fastlanes.bitpacked.v2(u16, len=16384) nbytes=6144 metadata: bit_width: 3, offset: 0 - right_parts: fastlanes.bitpacked(u64, len=16384) nbytes=106496 + right_parts: fastlanes.bitpacked.v2(u64, len=16384) nbytes=106496 metadata: bit_width: 52, offset: 0 patch_indices: vortex.primitive(u16, len=60) nbytes=120 metadata: ptype: u16 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__float_low_cardinality.snap b/vortex-btrblocks/tests/snapshots/golden__regular__float_low_cardinality.snap index 9c2192c70d7..59291fba1e2 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__float_low_cardinality.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__float_low_cardinality.snap @@ -7,7 +7,7 @@ root: vortex.alp(f64, len=16384) nbytes=6208 metadata: exponents: e: 16, f: 11 encoded: vortex.dict(i64, len=16384) nbytes=6208 metadata: all_values_referenced: true - codes: fastlanes.bitpacked(u8, len=16384) nbytes=6144 + codes: fastlanes.bitpacked.v2(u8, len=16384) nbytes=6144 metadata: bit_width: 3, offset: 0 values: vortex.primitive(i64, len=8) nbytes=64 metadata: ptype: i64 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__int_low_cardinality.snap b/vortex-btrblocks/tests/snapshots/golden__regular__int_low_cardinality.snap index 8648200a138..a6d01e4f5ea 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__int_low_cardinality.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__int_low_cardinality.snap @@ -5,7 +5,7 @@ expression: rendered input: i64, len=16384, nbytes=131072 root: vortex.dict(i64, len=16384) nbytes=6192 metadata: all_values_referenced: true - codes: fastlanes.bitpacked(u8, len=16384) nbytes=6144 + codes: fastlanes.bitpacked.v2(u8, len=16384) nbytes=6144 metadata: bit_width: 3, offset: 0 values: vortex.primitive(i64, len=6) nbytes=48 metadata: ptype: i64 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__int_monotone_jitter.snap b/vortex-btrblocks/tests/snapshots/golden__regular__int_monotone_jitter.snap index 47fd238df48..eb7d562744c 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__int_monotone_jitter.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__int_monotone_jitter.snap @@ -5,7 +5,7 @@ expression: rendered input: u64, len=16384, nbytes=131072 root: fastlanes.for.v2(u64, len=16384) nbytes=49159 metadata: offset: 0 - encoded: fastlanes.bitpacked(u64, len=16384) nbytes=49152 + encoded: fastlanes.bitpacked.v2(u64, len=16384) nbytes=49152 metadata: bit_width: 24, offset: 0 references: vortex.constant(u64, len=16) nbytes=7 metadata: scalar: 1700000001036u64 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__int_mostly_null.snap b/vortex-btrblocks/tests/snapshots/golden__regular__int_mostly_null.snap index 73566369279..2660f322248 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__int_mostly_null.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__int_mostly_null.snap @@ -7,7 +7,7 @@ root: vortex.sparse(i32?, len=16384) nbytes=3031 metadata: fill_value: null patch_indices: vortex.primitive(u16, len=823) nbytes=1646 metadata: ptype: u16 - patch_values: fastlanes.bitpacked(i32?, len=823) nbytes=1383 + patch_values: fastlanes.bitpacked.v2(i32?, len=823) nbytes=1383 metadata: bit_width: 10, offset: 0 validity_child: vortex.bool(bool, len=823) nbytes=103 metadata: offset: 0 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__int_negatives.snap b/vortex-btrblocks/tests/snapshots/golden__regular__int_negatives.snap index 71969bd5250..82c896b6082 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__int_negatives.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__int_negatives.snap @@ -5,7 +5,7 @@ expression: rendered input: i64, len=16384, nbytes=131072 root: fastlanes.for.v2(i64, len=16384) nbytes=16387 metadata: offset: 0 - encoded: fastlanes.bitpacked(i64, len=16384) nbytes=16384 + encoded: fastlanes.bitpacked.v2(i64, len=16384) nbytes=16384 metadata: bit_width: 8, offset: 0 references: vortex.constant(i64, len=16) nbytes=3 metadata: scalar: -128i64 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__int_runs.snap b/vortex-btrblocks/tests/snapshots/golden__regular__int_runs.snap index 17879983065..84e86906f6d 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__int_runs.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__int_runs.snap @@ -7,13 +7,13 @@ root: vortex.runend(i32, len=16384) nbytes=3974 metadata: offset: 0 ends: fastlanes.for.v2(u16, len=1020) nbytes=1794 metadata: offset: 0 - encoded: fastlanes.bitpacked(u16, len=1020) nbytes=1792 + encoded: fastlanes.bitpacked.v2(u16, len=1020) nbytes=1792 metadata: bit_width: 14, offset: 0 references: vortex.constant(u16, len=1) nbytes=2 metadata: scalar: 13u16 values: fastlanes.for.v2(i32, len=1020) nbytes=2180 metadata: offset: 0 - encoded: fastlanes.bitpacked(i32, len=1020) nbytes=2176 + encoded: fastlanes.bitpacked.v2(i32, len=1020) nbytes=2176 metadata: bit_width: 17, offset: 0 references: vortex.constant(i32, len=1) nbytes=4 metadata: scalar: -49931i32 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__int_sparse_outliers.snap b/vortex-btrblocks/tests/snapshots/golden__regular__int_sparse_outliers.snap index b4de2764333..38ca753c9ad 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__int_sparse_outliers.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__int_sparse_outliers.snap @@ -9,7 +9,7 @@ root: vortex.sparse(i64, len=16384) nbytes=5546 metadata: ptype: u16 patch_values: fastlanes.for.v2(i64, len=848) nbytes=3846 metadata: offset: 0 - encoded: fastlanes.bitpacked(i64, len=848) nbytes=3840 + encoded: fastlanes.bitpacked.v2(i64, len=848) nbytes=3840 metadata: bit_width: 30, offset: 0 references: vortex.constant(i64, len=1) nbytes=6 metadata: scalar: 1000830099i64 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__list_of_int_runs.snap b/vortex-btrblocks/tests/snapshots/golden__regular__list_of_int_runs.snap index 9ea63e9f0c6..612518768c6 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__list_of_int_runs.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__list_of_int_runs.snap @@ -9,17 +9,17 @@ root: vortex.list(list(i32), len=4066) nbytes=11152 metadata: offset: 0 ends: fastlanes.for.v2(u16, len=1020) nbytes=1794 metadata: offset: 0 - encoded: fastlanes.bitpacked(u16, len=1020) nbytes=1792 + encoded: fastlanes.bitpacked.v2(u16, len=1020) nbytes=1792 metadata: bit_width: 14, offset: 0 references: vortex.constant(u16, len=1) nbytes=2 metadata: scalar: 13u16 values: fastlanes.for.v2(i32, len=1020) nbytes=2180 metadata: offset: 0 - encoded: fastlanes.bitpacked(i32, len=1020) nbytes=2176 + encoded: fastlanes.bitpacked.v2(i32, len=1020) nbytes=2176 metadata: bit_width: 17, offset: 0 references: vortex.constant(i32, len=1) nbytes=4 metadata: scalar: -49931i32 - offsets: fastlanes.bitpacked(u16, len=4067) nbytes=7178 + offsets: fastlanes.bitpacked.v2(u16, len=4067) nbytes=7178 metadata: bit_width: 14, offset: 0 patch_indices: vortex.primitive(u16, len=1) nbytes=2 metadata: ptype: u16 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__string_fsst_structured.snap b/vortex-btrblocks/tests/snapshots/golden__regular__string_fsst_structured.snap index 327f050b0d7..20c4e76deb4 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__string_fsst_structured.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__string_fsst_structured.snap @@ -11,5 +11,5 @@ root: vortex.fsst(utf8, len=16384) nbytes=151382 metadata: ptype: u16 patch_values: vortex.constant(u8, len=1575) nbytes=2 metadata: scalar: 23u8 - codes_offsets: fastlanes.bitpacked(u32, len=16385) nbytes=36992 + codes_offsets: fastlanes.bitpacked.v2(u32, len=16385) nbytes=36992 metadata: bit_width: 17, offset: 0 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__string_low_cardinality.snap b/vortex-btrblocks/tests/snapshots/golden__regular__string_low_cardinality.snap index ec79989d46a..6217402155d 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__string_low_cardinality.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__string_low_cardinality.snap @@ -5,7 +5,7 @@ expression: rendered input: utf8, len=16384, nbytes=262144 root: vortex.dict(utf8, len=16384) nbytes=8373 metadata: all_values_referenced: true - codes: fastlanes.bitpacked(u8, len=16384) nbytes=8192 + codes: fastlanes.bitpacked.v2(u8, len=16384) nbytes=8192 metadata: bit_width: 4, offset: 0 values: vortex.fsst(utf8, len=12) nbytes=181 metadata: len: 12, nsymbols: 9 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__struct_mixed.snap b/vortex-btrblocks/tests/snapshots/golden__regular__struct_mixed.snap index be8f48b779b..81f37902b41 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__struct_mixed.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__struct_mixed.snap @@ -9,7 +9,7 @@ root: vortex.struct({id=i64, category=utf8, value=f64}, len=16384) nbytes=57525 metadata: base: 10000i64, multiplier: 7i64 category: vortex.dict(utf8, len=16384) nbytes=8373 metadata: all_values_referenced: true - codes: fastlanes.bitpacked(u8, len=16384) nbytes=8192 + codes: fastlanes.bitpacked.v2(u8, len=16384) nbytes=8192 metadata: bit_width: 4, offset: 0 values: vortex.fsst(utf8, len=12) nbytes=181 metadata: len: 12, nsymbols: 9 @@ -19,5 +19,5 @@ root: vortex.struct({id=i64, category=utf8, value=f64}, len=16384) nbytes=57525 metadata: ptype: u8 value: vortex.alp(f64, len=16384) nbytes=49152 metadata: exponents: e: 14, f: 12 - encoded: fastlanes.bitpacked(i64, len=16384) nbytes=49152 + encoded: fastlanes.bitpacked.v2(i64, len=16384) nbytes=49152 metadata: bit_width: 24, offset: 0 diff --git a/vortex-btrblocks/tests/snapshots/golden__regular__temporal_timestamp_micros.snap b/vortex-btrblocks/tests/snapshots/golden__regular__temporal_timestamp_micros.snap index 34339fc8d9a..5642e0c8bcf 100644 --- a/vortex-btrblocks/tests/snapshots/golden__regular__temporal_timestamp_micros.snap +++ b/vortex-btrblocks/tests/snapshots/golden__regular__temporal_timestamp_micros.snap @@ -7,7 +7,7 @@ root: vortex.ext(vortex.timestamp[µs, tz=UTC](i64), len=16384) nbytes=67593 metadata: storage: fastlanes.for.v2(i64, len=16384) nbytes=67593 metadata: offset: 0 - encoded: fastlanes.bitpacked(i64, len=16384) nbytes=67584 + encoded: fastlanes.bitpacked.v2(i64, len=16384) nbytes=67584 metadata: bit_width: 33, offset: 0 references: vortex.constant(i64, len=16) nbytes=9 metadata: scalar: 1700000000891673i64 diff --git a/vortex-cuda/benches/dynamic_dispatch_cuda.rs b/vortex-cuda/benches/dynamic_dispatch_cuda.rs index 82afa0a843d..da456fb9d31 100644 --- a/vortex-cuda/benches/dynamic_dispatch_cuda.rs +++ b/vortex-cuda/benches/dynamic_dispatch_cuda.rs @@ -47,6 +47,7 @@ use vortex::encodings::alp::alp_encode; use vortex::encodings::fastlanes::BitPackedArray; use vortex::encodings::fastlanes::BitPackedArrayExt; use vortex::encodings::fastlanes::BitPackedData; +use vortex::encodings::fastlanes::BitWidths; use vortex::encodings::fastlanes::FoR; use vortex::encodings::fastlanes::FoRArrayExt; use vortex::encodings::fastlanes::FoRArraySlotsExt; @@ -475,8 +476,8 @@ mod standalone { cuda_session: &CudaSession, cuda_ctx: &mut CudaExecutionCtx, ) -> Self { - assert_eq!(values_bp.bit_width(), 6); - assert_eq!(codes_bp.bit_width(), 6); + assert!(matches!(values_bp.bit_widths(), BitWidths::Global(6))); + assert!(matches!(codes_bp.bit_widths(), BitWidths::Global(6))); let values_packed = block_on(cuda_ctx.ensure_on_device(values_bp.packed().clone())) .vortex_expect("values packed"); diff --git a/vortex-cuda/src/dynamic_dispatch/plan_builder.rs b/vortex-cuda/src/dynamic_dispatch/plan_builder.rs index bd143116cdd..0ddc08dcdfb 100644 --- a/vortex-cuda/src/dynamic_dispatch/plan_builder.rs +++ b/vortex-cuda/src/dynamic_dispatch/plan_builder.rs @@ -30,6 +30,7 @@ use vortex::encodings::alp::ALPFloat; use vortex::encodings::alp::Exponents; use vortex::encodings::fastlanes::BitPacked; use vortex::encodings::fastlanes::BitPackedArrayExt; +use vortex::encodings::fastlanes::BitWidths; use vortex::encodings::fastlanes::FoR; use vortex::encodings::fastlanes::FoRArrayExt; use vortex::encodings::fastlanes::FoRArraySlotsExt; @@ -89,7 +90,7 @@ fn is_dyn_dispatch_compatible(array: &ArrayRef) -> bool { return matches!(arr.dtype().as_ptype(), PType::F32 | PType::F64); } if id == BitPacked.id() { - return true; + return is_bitpacked_with_global_bit_width(array); } if id == Dict.id() { let arr = array.as_::(); @@ -156,11 +157,16 @@ fn is_dyn_dispatch_cast_compatible(array: &ArrayRef) -> bool { /// Returns `true` if a registered standalone kernel can decode the entire /// `array` tree in a single launch without recursing into `execute_cuda` /// for child encodings. +/// +/// `FoR` requires a constant reference, and `BitPacked` requires a constant bit +/// width. pub fn has_standalone_kernel(array: &ArrayRef) -> bool { let id = array.encoding_id(); - // Leaf encodings: no children to recurse into. - if id == BitPacked.id() || id == Sequence.id() { + if id == BitPacked.id() { + return is_bitpacked_with_global_bit_width(array); + } + if id == Sequence.id() { return true; } @@ -172,10 +178,10 @@ pub fn has_standalone_kernel(array: &ArrayRef) -> bool { } let child = for_arr.encoded(); if child.encoding_id() == BitPacked.id() { - return true; + return is_bitpacked_with_global_bit_width(child); } if let Some(slice) = child.as_opt::() { - return slice.child().encoding_id() == BitPacked.id(); + return is_bitpacked_with_global_bit_width(slice.child()); } return false; } @@ -183,6 +189,12 @@ pub fn has_standalone_kernel(array: &ArrayRef) -> bool { false } +fn is_bitpacked_with_global_bit_width(array: &ArrayRef) -> bool { + array + .as_opt::() + .is_some_and(|array| array.bit_widths().is_global()) +} + /// Patch payload attached to the op that consumes it. /// /// `range` is the logical output range to apply when materializing the patch descriptor on the GPU. @@ -562,7 +574,8 @@ impl FusedPlan { let bp = child.as_::(); let offset = slice_arr.data().slice_range().start; let len = array.len(); - let (packed, bitpacked_offset, patch_range) = bitpacked_slice_view(bp, offset, len)?; + let (packed, bit_width, bitpacked_offset, patch_range) = + bitpacked_slice_view(bp, offset, len)?; let source_ptype = ptype_to_tag(PType::try_from(bp.dtype()).map_err(|_| { vortex_err!("BitPacked must have primitive dtype, got {:?}", bp.dtype()) @@ -570,7 +583,7 @@ impl FusedPlan { let buf_index = self.source_buffers.len(); self.source_buffers.push(Some(packed)); return Ok(Stage::new( - SourceOp::bitunpack(bp.bit_width(), bitpacked_offset), + SourceOp::bitunpack(bit_width, bitpacked_offset), Some(buf_index), source_ptype, ) @@ -622,10 +635,13 @@ impl FusedPlan { let source_ptype = ptype_to_tag(PType::try_from(bp.dtype()).map_err(|_| { vortex_err!("BitPacked must have primitive dtype, got {:?}", bp.dtype()) })?); + let BitWidths::Global(bit_width) = bp.bit_widths() else { + vortex_bail!("CUDA does not support BitPacked arrays with per-block bit widths"); + }; let buf_index = self.source_buffers.len(); self.source_buffers.push(Some(bp.packed().clone())); Ok(Stage::new( - SourceOp::bitunpack(bp.bit_width(), bp.offset()), + SourceOp::bitunpack(bit_width, bp.offset()), Some(buf_index), source_ptype, ) @@ -903,14 +919,50 @@ impl FusedPlan { #[cfg(test)] mod tests { + use rstest::rstest; use vortex::array::IntoArray; use vortex::array::arrays::PrimitiveArray; + use vortex::array::arrays::SliceArray; use vortex::array::builtins::ArrayBuiltins; + use vortex::buffer::Buffer; + use vortex::buffer::ByteBuffer; + use vortex::buffer::buffer; use vortex::dtype::DType; use vortex::dtype::Nullability; use super::*; + #[rstest] + #[case::equal_steps(buffer![0u64, 512, 1024])] + #[case::different_widths(buffer![0u64, 384, 1024])] + fn materialized_bitpacked_offsets_have_no_standalone_kernel( + #[case] offsets: Buffer, + ) -> VortexResult<()> { + let bitpacked = BitPacked::try_new_with_block_offsets( + BufferHandle::new_host(ByteBuffer::zeroed(1024)), + PType::U32, + Validity::NonNullable, + None, + offsets.into_array(), + 2048, + 0, + )? + .into_array(); + assert!(!has_standalone_kernel(&bitpacked)); + assert!(matches!( + DispatchPlan::new(&bitpacked, CudaDispatchMode::Auto)?, + DispatchPlan::Unfused + )); + + let for_bitpacked = FoR::try_new(bitpacked.clone(), 100u32.into())?.into_array(); + assert!(!has_standalone_kernel(&for_bitpacked)); + + let sliced = SliceArray::new(bitpacked, 100..1500).into_array(); + let for_sliced = FoR::try_new(sliced, 100u32.into())?.into_array(); + assert!(!has_standalone_kernel(&for_sliced)); + Ok(()) + } + #[test] fn cast_to_non_primitive_target_is_not_dyn_dispatch_compatible() -> VortexResult<()> { let cast = PrimitiveArray::from_iter([0u8, 1]) diff --git a/vortex-cuda/src/kernel/encodings/bitpacked.rs b/vortex-cuda/src/kernel/encodings/bitpacked.rs index 86b7a88b276..9db84b97d7f 100644 --- a/vortex-cuda/src/kernel/encodings/bitpacked.rs +++ b/vortex-cuda/src/kernel/encodings/bitpacked.rs @@ -26,8 +26,10 @@ use vortex::encodings::fastlanes::BitPacked; use vortex::encodings::fastlanes::BitPackedArray; use vortex::encodings::fastlanes::BitPackedArrayExt; use vortex::encodings::fastlanes::BitPackedDataParts; +use vortex::encodings::fastlanes::BitWidths; use vortex::encodings::fastlanes::unpack_iter::BitPacked as BitPackedUnpack; use vortex::error::VortexResult; +use vortex::error::vortex_bail; use vortex::error::vortex_ensure; use vortex::error::vortex_err; @@ -48,12 +50,13 @@ pub(crate) struct BitPackedExecutor; /// Bit-unpack kernels decode full FastLanes chunks, so the packed buffer is /// widened to chunk boundaries and `offset` is converted into the in-chunk /// starting position. The returned logical range is passed to patch -/// materialization so exception metadata is sliced consistently. +/// materialization so exception metadata is sliced consistently. The global +/// bit width of `bp` is returned alongside the view. pub(crate) fn bitpacked_slice_view( bp: ArrayView<'_, BitPacked>, offset: usize, len: usize, -) -> VortexResult<(BufferHandle, u16, Range)> { +) -> VortexResult<(BufferHandle, u8, u16, Range)> { let patch_range = offset..offset + len; let offset_start = patch_range.start + bp.offset() as usize; let offset_stop = offset_start + len; @@ -61,11 +64,15 @@ pub(crate) fn bitpacked_slice_view( let block_start = offset_start - bitpacked_offset; let block_stop = offset_stop.div_ceil(PATCH_CHUNK_SIZE) * PATCH_CHUNK_SIZE; - let encoded_start = (block_start / 8) * bp.bit_width() as usize; - let encoded_stop = (block_stop / 8) * bp.bit_width() as usize; + let BitWidths::Global(bit_width) = bp.bit_widths() else { + vortex_bail!("CUDA does not support BitPacked arrays with per-block bit widths"); + }; + let encoded_start = (block_start / 8) * bit_width as usize; + let encoded_stop = (block_stop / 8) * bit_width as usize; Ok(( bp.packed().slice(encoded_start..encoded_stop), + bit_width, u16::try_from(bitpacked_offset)?, patch_range, )) @@ -90,13 +97,14 @@ impl BitPackedExecutor { let bp = child.as_::(); let offset = slice.data().slice_range().start; let len = array.len(); - let (packed, bitpacked_offset, patch_range) = bitpacked_slice_view(bp, offset, len)?; + let (packed, bit_width, bitpacked_offset, patch_range) = + bitpacked_slice_view(bp, offset, len)?; let sliced = BitPacked::try_new( packed, bp.ptype(bp.dtype()), child.validity()?.slice(patch_range.clone())?, bp.patches(), - bp.bit_width(), + bit_width, len, bitpacked_offset, )?; @@ -162,12 +170,15 @@ where { let BitPackedDataParts { offset, - bit_width, + bit_widths, len, packed, patches, validity, } = BitPacked::into_parts(array); + let BitWidths::Global(bit_width) = bit_widths else { + vortex_bail!("CUDA does not support BitPacked arrays with per-block bit widths"); + }; vortex_ensure!(len > 0, "Non empty array"); let offset = offset as usize; diff --git a/vortex-cuda/src/kernel/encodings/for_.rs b/vortex-cuda/src/kernel/encodings/for_.rs index 694f7fe1902..a5e420939bf 100644 --- a/vortex-cuda/src/kernel/encodings/for_.rs +++ b/vortex-cuda/src/kernel/encodings/for_.rs @@ -20,6 +20,7 @@ use vortex::array::match_each_integer_ptype; use vortex::array::match_each_native_simd_ptype; use vortex::dtype::NativePType; use vortex::encodings::fastlanes::BitPacked; +use vortex::encodings::fastlanes::BitPackedArrayExt; use vortex::encodings::fastlanes::FoR; use vortex::encodings::fastlanes::FoRArray; use vortex::encodings::fastlanes::FoRArrayExt; @@ -65,7 +66,9 @@ impl CudaExecute for FoRExecutor { }; // Fuse FOR + BP => FFOR - if let Some(bitpacked) = array.encoded().as_opt::() { + if let Some(bitpacked) = array.encoded().as_opt::() + && bitpacked.bit_widths().is_global() + { match_each_integer_ptype!(bitpacked.ptype(bitpacked.dtype()), |P| { let reference: P = (&reference).try_into()?; return decode_bitpacked(bitpacked.into_owned(), reference, None, ctx).await; @@ -75,6 +78,7 @@ impl CudaExecute for FoRExecutor { // Fuse FOR + SLICE + BP => SLICE + FFOR if let Some(slice_array) = array.encoded().as_opt::() && let Some(bitpacked) = slice_array.child().as_opt::() + && bitpacked.bit_widths().is_global() { let slice_range = slice_array.slice_range().clone(); let unpacked = match_each_integer_ptype!(bitpacked.ptype(bitpacked.dtype()), |P| { diff --git a/vortex-file/src/tests.rs b/vortex-file/src/tests.rs index 6f74e583b7c..e2d1183398d 100644 --- a/vortex-file/src/tests.rs +++ b/vortex-file/src/tests.rs @@ -1537,8 +1537,9 @@ async fn test_into_tokio_array_stream() -> VortexResult<()> { async fn test_array_stream_no_double_dict_encode() -> VortexResult<()> { let num_vals = 2048; let mut values = Vec::::with_capacity(num_vals); - values.extend(iter::repeat_n(0, num_vals / 2)); - values.extend(iter::repeat_n(1, num_vals / 2)); + // Far apart, so per-block bit widths can't pack them tighter than a dictionary. + values.extend(iter::repeat_n(1_000_000_000, num_vals / 2)); + values.extend(iter::repeat_n(2_000_000_000, num_vals / 2)); let array = PrimitiveArray::from_iter(values).into_array(); let mut buf = Vec::new(); diff --git a/vortex-python/python/vortex/_lib/arrays.pyi b/vortex-python/python/vortex/_lib/arrays.pyi index a0c8317e9db..bc36d6b72ef 100644 --- a/vortex-python/python/vortex/_lib/arrays.pyi +++ b/vortex-python/python/vortex/_lib/arrays.pyi @@ -133,7 +133,7 @@ class ZigZagArray(Array): @final class FastLanesBitPackedArray(Array): @property - def bit_width(self) -> int: ... + def bit_width(self) -> int | None: ... @final class FastLanesDeltaArray(Array): ... diff --git a/vortex-python/src/arrays/fastlanes.rs b/vortex-python/src/arrays/fastlanes.rs index 31b49e8d804..8b8b9cc0b03 100644 --- a/vortex-python/src/arrays/fastlanes.rs +++ b/vortex-python/src/arrays/fastlanes.rs @@ -3,10 +3,11 @@ use pyo3::prelude::*; use vortex::encodings::fastlanes::BitPacked; +use vortex::encodings::fastlanes::BitPackedArrayExt; +use vortex::encodings::fastlanes::BitWidths; use vortex::encodings::fastlanes::Delta; use vortex::encodings::fastlanes::FoR; -use crate::arrays::native::AsArrayRef; use crate::arrays::native::EncodingSubclass; use crate::arrays::native::PyNativeArray; @@ -20,10 +21,14 @@ impl EncodingSubclass for PyFastLanesBitPackedArray { #[pymethods] impl PyFastLanesBitPackedArray { - /// Returns the bit width of the packed values. + /// Returns the global bit width of the packed values, or `None` if the array has per-block + /// bit widths. #[getter] - fn bit_width(self_: PyRef<'_, Self>) -> u8 { - self_.as_array_ref().bit_width() + fn bit_width(self_: PyRef<'_, Self>) -> Option { + match self_.as_super().inner().as_::().bit_widths() { + BitWidths::Global(bit_width) => Some(bit_width), + BitWidths::Blocked(_) => None, + } } }