Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 4 additions & 3 deletions vortex-btrblocks/src/builder.rs
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@

use vortex_array::ArrayId;
use vortex_decimal_byte_parts::decimal_byte_parts_v2_id;
use vortex_fastlanes::bitpacked_v2_id;
use vortex_fastlanes::for_v2_id;
use vortex_session::VortexSession;
use vortex_utils::aliases::hash_set::HashSet;
Expand Down Expand Up @@ -94,9 +95,9 @@ impl CompressionMode {
fn excluded_encodings(self) -> Vec<ArrayId> {
match self {
Self::All | Self::Default | Self::Compact => Vec::new(),
// Multi-part DecimalByteParts arrays and FoR arrays with per-chunk references have no
// CUDA decode kernel.
Self::Cuda => vec![decimal_byte_parts_v2_id(), for_v2_id()],
// Multi-part DecimalByteParts arrays, FoR arrays with per-chunk references and
// BitPacked arrays with per-block bit widths have no CUDA decode kernel.
Self::Cuda => vec![decimal_byte_parts_v2_id(), for_v2_id(), bitpacked_v2_id()],
}
}
}
Expand Down
184 changes: 179 additions & 5 deletions vortex-btrblocks/src/schemes/integer/bitpacking.rs
Original file line number Diff line number Diff line change
Expand Up @@ -16,21 +16,69 @@ use vortex_compressor::scheme::CompressionEstimate;
use vortex_compressor::scheme::DeferredEstimate;
use vortex_compressor::scheme::EstimateVerdict;
use vortex_error::VortexResult;
use vortex_error::vortex_bail;
use vortex_fastlanes::BitPacked;
use vortex_fastlanes::BitWidths;
use vortex_fastlanes::bitpack_compress::bit_width_histogram;
use vortex_fastlanes::bitpack_compress::bitpack_blocked_to_best_bit_widths;
use vortex_fastlanes::bitpack_compress::bitpack_encode;
use vortex_fastlanes::bitpack_compress::find_best_bit_width;
use vortex_fastlanes::bitpacked_v1_id;
use vortex_fastlanes::bitpacked_v2_id;

use crate::ArrayAndStats;
use crate::CascadingCompressor;
use crate::CompressorContext;
use crate::Scheme;
use crate::SchemeExt;
use crate::compress_patches;

#[derive(Debug, Copy, Clone, PartialEq, Eq)]
enum BitPackingSchemeMode {
V1,
V2,
}

/// The BitPacking scheme that only produces arrays with one bit width.
pub(crate) static BITPACKING_V1: BitPackingScheme = BitPackingScheme::v1();

/// The BitPacking scheme that may give each 1024-element block its own bit width.
pub(crate) static BITPACKING_V2: BitPackingScheme = BitPackingScheme::v2();

/// BitPacking encoding for non-negative integers.
///
/// The v1 mode packs every value at one bit width, and serializes as `fastlanes.bitpacked`. The v2
/// mode always chooses a width for each 1024-element block, and serializes as
/// `fastlanes.bitpacked.v2`.
///
/// The default uses v1. [`refine`](Scheme::refine) picks v2 when the v2 ID is allowed and v1
/// otherwise.
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
pub struct BitPackingScheme;
pub struct BitPackingScheme {
mode: BitPackingSchemeMode,
}

impl BitPackingScheme {
/// Creates a BitPacking scheme configured for v1, which uses one bit width.
pub const fn v1() -> Self {
Self {
mode: BitPackingSchemeMode::V1,
}
}

/// Creates a BitPacking scheme configured for v2, which may use one bit width per block.
pub const fn v2() -> Self {
Self {
mode: BitPackingSchemeMode::V2,
}
}
}

impl Default for BitPackingScheme {
fn default() -> Self {
Self::v1()
}
}

impl Scheme for BitPackingScheme {
fn scheme_name(&self) -> &'static str {
Expand All @@ -42,14 +90,33 @@ impl Scheme for BitPackingScheme {
}

fn produced_encodings(&self) -> Vec<ArrayId> {
// Global-width arrays serialize under the frozen v1 ID.
let mut encodings = vec![bitpacked_v1_id()];
let mut encodings = match self.mode {
// Global-width arrays serialize under the frozen v1 ID.
BitPackingSchemeMode::V1 => vec![bitpacked_v1_id()],
BitPackingSchemeMode::V2 => vec![bitpacked_v2_id()],
};
if use_experimental_patches() {
encodings.push(Patched.id());
}
encodings
}

fn refine(&self, allowed: &dyn Fn(&ArrayId) -> bool) -> &dyn Scheme {
if allowed(&bitpacked_v2_id()) {
&BITPACKING_V2
} else {
&BITPACKING_V1
}
}

/// Children: block offsets=0 in v2 mode.
fn num_children(&self) -> usize {
match self.mode {
BitPackingSchemeMode::V1 => 0,
BitPackingSchemeMode::V2 => 1,
}
}

fn expected_compression_ratio(
&self,
data: &ArrayAndStats,
Expand All @@ -68,11 +135,15 @@ impl Scheme for BitPackingScheme {

fn compress(
&self,
_compressor: &CascadingCompressor,
compressor: &CascadingCompressor,
data: &ArrayAndStats,
_compress_ctx: CompressorContext,
compress_ctx: CompressorContext,
exec_ctx: &mut ExecutionCtx,
) -> VortexResult<ArrayRef> {
if self.mode == BitPackingSchemeMode::V2 {
return self.compress_blocked(compressor, data, compress_ctx, exec_ctx);
}

let primitive_array = data.array_as_primitive();

let histogram = bit_width_histogram(primitive_array, exec_ctx)?;
Expand Down Expand Up @@ -135,3 +206,106 @@ impl Scheme for BitPackingScheme {
Ok(array)
}
}

impl BitPackingScheme {
/// Bit-pack each 1024-element block at its own best width.
fn compress_blocked(
&self,
compressor: &CascadingCompressor,
data: &ArrayAndStats,
compress_ctx: CompressorContext,
exec_ctx: &mut ExecutionCtx,
) -> VortexResult<ArrayRef> {
let primitive_array = data.array_as_primitive().into_owned();
let packed = bitpack_blocked_to_best_bit_widths(&primitive_array, exec_ctx)?;

let packed_stats = packed.statistics().to_owned();
let ptype = packed.dtype().as_ptype();
let mut parts = BitPacked::into_parts(packed);
let BitWidths::Blocked(block_offsets) = parts.bit_widths else {
vortex_bail!("Blocked bit-packing must produce block offsets");
};
let block_offsets =
compressor.compress_child(&block_offsets, &compress_ctx, self.id(), 0, exec_ctx)?;

let array = if use_experimental_patches() {
let patches = parts.patches.take();
// Transpose patches into G-ALP style PatchedArray, wrapping an inner BitPackedArray.
let array = BitPacked::try_new_with_block_offsets(
parts.packed,
ptype,
parts.validity,
None,
block_offsets,
parts.len,
parts.offset,
)?
.into_array();

match patches {
None => array,
Some(p) => Patched::from_array_and_patches(array, &p, exec_ctx)?
.with_stats_set(packed_stats)
.into_array(),
}
} else {
// Compress patches and place back into BitPackedArray.
let patches = parts
.patches
.take()
.map(|p| compress_patches(p, exec_ctx))
.transpose()?;
BitPacked::try_new_with_block_offsets(
parts.packed,
ptype,
parts.validity,
patches,
block_offsets,
parts.len,
parts.offset,
)?
.with_stats_set(packed_stats)
.into_array()
};

Ok(array)
}
}

#[cfg(test)]
mod tests {
use rstest::rstest;
use vortex_array::ArrayId;
use vortex_fastlanes::bitpacked_v1_id;
use vortex_fastlanes::bitpacked_v2_id;

use super::BITPACKING_V1;
use super::BITPACKING_V2;
use crate::Scheme;
use crate::SchemeExt;

/// Both variants refine to v2 exactly when the v2 ID is allowed.
#[rstest]
#[case::neither(false, false, false)]
#[case::v1(true, false, false)]
#[case::v2_only(false, true, true)]
#[case::both(true, true, true)]
fn refine_picks_v2_when_v2_id_is_allowed(
#[case] allow_v1: bool,
#[case] allow_v2: bool,
#[case] expect_v2: bool,
#[values(&BITPACKING_V1, &BITPACKING_V2)] scheme: &'static dyn Scheme,
) {
let allowed = |id: &ArrayId| {
(allow_v1 && *id == bitpacked_v1_id()) || (allow_v2 && *id == bitpacked_v2_id())
};
let refined = scheme.refine(&allowed);
assert_eq!(refined.id(), scheme.id());
let expected: &dyn Scheme = if expect_v2 {
&BITPACKING_V2
} else {
&BITPACKING_V1
};
assert_eq!(refined.produced_encodings(), expected.produced_encodings());
}
}
4 changes: 2 additions & 2 deletions vortex-btrblocks/src/schemes/integer/for_.rs
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,7 @@ use vortex_fastlanes::FoRArraySlotsExt;
use vortex_fastlanes::for_v1_id;
use vortex_fastlanes::for_v2_id;

use super::BitPackingScheme;
use super::BITPACKING_V1;
use crate::ArrayAndStats;
use crate::CascadingCompressor;
use crate::CompressorContext;
Expand Down Expand Up @@ -216,7 +216,7 @@ impl Scheme for FoRScheme {
let leaf_ctx = compress_ctx.clone().as_leaf();
let biased_data =
ArrayAndStats::new(biased.into_array(), compress_ctx.merged_stats_options());
let compressed = BitPackingScheme.compress(compressor, &biased_data, leaf_ctx, exec_ctx)?;
let compressed = BITPACKING_V1.compress(compressor, &biased_data, leaf_ctx, exec_ctx)?;

// TODO(connor): This should really be `new_unchecked`.
let for_compressed = match for_array.constant_reference() {
Expand Down
1 change: 1 addition & 0 deletions vortex-btrblocks/src/schemes/integer/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@ mod zigzag;
#[cfg(feature = "pco")]
mod pco;

pub(crate) use bitpacking::BITPACKING_V1;
pub use bitpacking::BitPackingScheme;
pub use delta::DeltaScheme;
pub(crate) use for_::FOR_V1;
Expand Down
2 changes: 1 addition & 1 deletion vortex-btrblocks/src/session.rs
Original file line number Diff line number Diff line change
Expand Up @@ -48,7 +48,7 @@ impl Default for CompressionSession {
&integer::FOR_V1,
// NOTE: ZigZag should precede BitPacking because we don't want negative numbers.
&integer::ZigZagScheme,
&integer::BitPackingScheme,
&integer::BITPACKING_V1,
&integer::SparseScheme,
&integer::IntDictScheme,
&integer::RunEndScheme,
Expand Down
Loading
Loading