11// SPDX-License-Identifier: Apache-2.0
22// SPDX-FileCopyrightText: Copyright the Vortex contributors
33
4- //! AVX-512 `vpcompress`, AVX2 `vpermd`, and 128-bit `pshufb` compress kernels.
4+ //! AVX-512 `vpcompress` and AVX2 `vpermd` compress kernels. The 1- and 2-byte AVX2 paths use the
5+ //! portable kernels in [`generic`](super::generic).
56//!
67//! See the [module docs](super) for how these fit the shared dispatch.
78
8- use std:: arch:: x86_64:: __m128i;
9- use std:: arch:: x86_64:: _mm_loadl_epi64;
10- use std:: arch:: x86_64:: _mm_loadu_si128;
11- use std:: arch:: x86_64:: _mm_shuffle_epi8;
12- use std:: arch:: x86_64:: _mm_storel_epi64;
13- use std:: arch:: x86_64:: _mm_storeu_si128;
149use std:: arch:: x86_64:: _mm256_loadu_si256;
1510use std:: arch:: x86_64:: _mm256_maskload_epi32;
1611use std:: arch:: x86_64:: _mm256_maskload_epi64;
@@ -45,8 +40,8 @@ use super::super::slice::for_each_mask_word;
4540use super :: super :: slice:: low_bits_mask;
4641use super :: Kernel ;
4742use super :: bulk_copy;
48- use super :: compress_lut ;
49- use super :: compress_tail ;
43+ use super :: generic :: compress_generic_8 ;
44+ use super :: generic :: compress_generic_16 ;
5045
5146/// Choose the widest available kernel above its benchmarked density crossover.
5247///
@@ -59,8 +54,8 @@ pub(super) fn select_kernel<T, const IN_PLACE: bool>(mask: &MaskValues) -> Optio
5954 4 if avx512f ( ) => ( compress_avx512_epi32 :: < IN_PLACE > as Kernel , 0.25 ) ,
6055 8 if avx512f ( ) => ( compress_avx512_epi64 :: < IN_PLACE > as Kernel , 0.30 ) ,
6156 // AVX-512F without VBMI2 (e.g. Skylake-X) falls through to these too.
62- 1 if avx2 ( ) => ( compress_pshufb_epi8 :: < IN_PLACE > as Kernel , 0.15 ) ,
63- 2 if avx2 ( ) => ( compress_pshufb_epi16 :: < IN_PLACE > as Kernel , 0.25 ) ,
57+ 1 if avx2 ( ) => ( compress_generic_8 :: < IN_PLACE > as Kernel , 0.15 ) ,
58+ 2 if avx2 ( ) => ( compress_generic_16 :: < IN_PLACE > as Kernel , 0.25 ) ,
6459 4 if avx2 ( ) => ( compress_avx2_epi32 :: < IN_PLACE > as Kernel , 0.25 ) ,
6560 8 if avx2 ( ) => ( compress_avx2_epi64 :: < IN_PLACE > as Kernel , 0.45 ) ,
6661 _ => return None ,
@@ -451,126 +446,3 @@ avx2_compress_kernel!(
451446 maskload: _mm256_maskload_epi64,
452447 maskstore: _mm256_maskstore_epi64
453448) ;
454-
455- /// Byte-index rows for `pshufb`, which always indexes a full 16-byte register even though only
456- /// the low 8 (1-byte elements) or all 16 (2-byte elements) bytes hold lanes.
457- static SHUF_LUT_8 : [ [ u8 ; 16 ] ; 256 ] = compress_lut :: < 256 , 16 > ( 8 , 1 ) ;
458- static SHUF_LUT_16 : [ [ u8 ; 16 ] ; 256 ] = compress_lut :: < 256 , 16 > ( 8 , 2 ) ;
459-
460- /// Generate an AVX2 `pshufb` kernel for 1- or 2-byte elements.
461- macro_rules! pshufb_compress_kernel {
462- (
463- $word_fn: ident,
464- $walk_fn: ident, elem_size:
465- $elem_size: literal, idx_lut:
466- $idx_lut: ident, load:
467- $load: ident, store:
468- $store: ident
469- ) => {
470- /// # Safety
471- ///
472- /// The CPU must support AVX2 and the pointer contract of
473- /// [`filter_slice_by_bitmap`](super::filter_slice_by_bitmap) /
474- /// [`filter_slice_mut_by_bitmap`](super::filter_slice_mut_by_bitmap) must hold.
475- #[ expect(
476- clippy:: cast_possible_truncation,
477- reason = "deliberate submask narrowing"
478- ) ]
479- #[ target_feature( enable = "avx2" ) ]
480- #[ inline]
481- unsafe fn $word_fn<const IN_PLACE : bool >(
482- src: * const u8 ,
483- dst: * mut u8 ,
484- word: u64 ,
485- word_start: usize ,
486- word_len: usize ,
487- mut write_pos: usize ,
488- ) -> usize {
489- if word == 0 {
490- return write_pos;
491- }
492- if word == low_bits_mask( word_len) {
493- // SAFETY: forwarded from the caller contract.
494- unsafe {
495- bulk_copy:: <IN_PLACE >( src, dst, word_start, word_len, write_pos, $elem_size)
496- } ;
497- return write_pos + word_len;
498- }
499-
500- // Empty chunks still store garbage that the next chunk overwrites; branching here
501- // regresses masks near the density crossover.
502- let mut sub = 0 ;
503- while sub + 8 <= word_len {
504- let m = ( ( word >> sub) & low_bits_mask( 8 ) ) as usize ;
505- // SAFETY: the chunk holds 8 in-bounds source elements.
506- let chunk = unsafe { $load( src. add( ( word_start + sub) * $elem_size) . cast( ) ) } ;
507- // SAFETY: every LUT row is 16 bytes.
508- let idx = unsafe { _mm_loadu_si128( $idx_lut[ m] . as_ptr( ) . cast( ) ) } ;
509- // SAFETY: out-of-place output has vector slack. In-place, the store ends within
510- // the source chunk already loaded, and later stores overwrite trailing garbage.
511- unsafe {
512- $store(
513- dst. add( write_pos * $elem_size) . cast:: <__m128i>( ) ,
514- _mm_shuffle_epi8( chunk, idx) ,
515- )
516- } ;
517- write_pos += m. count_ones( ) as usize ;
518- sub += 8 ;
519- }
520-
521- if sub < word_len {
522- let bits = ( word >> sub) & low_bits_mask( word_len - sub) ;
523- // SAFETY: forwarded from the caller contract.
524- write_pos = unsafe {
525- compress_tail:: <IN_PLACE >(
526- src,
527- dst,
528- bits,
529- word_start + sub,
530- write_pos,
531- $elem_size,
532- )
533- } ;
534- }
535-
536- write_pos
537- }
538-
539- /// # Safety
540- ///
541- /// The CPU must support AVX2 and the pointer contract of
542- /// [`filter_slice_by_bitmap`](super::filter_slice_by_bitmap) /
543- /// [`filter_slice_mut_by_bitmap`](super::filter_slice_mut_by_bitmap) must hold.
544- #[ target_feature( enable = "avx2" ) ]
545- pub ( super ) unsafe fn $walk_fn<const IN_PLACE : bool >(
546- src: * const u8 ,
547- dst: * mut u8 ,
548- mask: & MaskValues ,
549- ) -> usize {
550- let mut write_pos = 0 ;
551- for_each_mask_word( mask, |word, word_start, word_len| {
552- // SAFETY: forwarded from the caller contract.
553- write_pos = unsafe {
554- $word_fn:: <IN_PLACE >( src, dst, word, word_start, word_len, write_pos)
555- } ;
556- } ) ;
557- write_pos
558- }
559- } ;
560- }
561-
562- pshufb_compress_kernel ! (
563- compress_word_pshufb_epi8, compress_pshufb_epi8,
564- elem_size: 1 ,
565- idx_lut: SHUF_LUT_8 ,
566- load: _mm_loadl_epi64,
567- store: _mm_storel_epi64
568- ) ;
569-
570- pshufb_compress_kernel ! (
571- compress_word_pshufb_epi16, compress_pshufb_epi16,
572- elem_size: 2 ,
573- idx_lut: SHUF_LUT_16 ,
574- load: _mm_loadu_si128,
575- store: _mm_storeu_si128
576- ) ;
0 commit comments