From c1109a7f0a868eaf31dc5a57b17eea91750c2379 Mon Sep 17 00:00:00 2001 From: Prateek Gaur Date: Sat, 3 Oct 2026 03:11:38 +0000 Subject: [PATCH 01/27] Fold frame bias into bit unpacking A frame-of-reference decoder adds the frame to every value right after unpacking it. Doing that in a second pass reads and writes the whole output again, so the scalar and SIMD unpackers now take an optional bias and add it before each store. With no bias the existing paths are unchanged, including the plain memcpy used when the bit width equals the output width. The tests run the biased form across widths, offsets, epilogues and every dispatched instruction set. --- cpp/src/arrow/util/bpacking.cc | 32 + .../arrow/util/bpacking_dispatch_internal.h | 602 ++++++++++-------- cpp/src/arrow/util/bpacking_internal.h | 25 + cpp/src/arrow/util/bpacking_scalar.cc | 15 + cpp/src/arrow/util/bpacking_scalar_internal.h | 16 + cpp/src/arrow/util/bpacking_simd_128.cc | 18 + cpp/src/arrow/util/bpacking_simd_128_alt.cc | 17 + cpp/src/arrow/util/bpacking_simd_256.cc | 18 + cpp/src/arrow/util/bpacking_simd_avx512.cc | 15 + cpp/src/arrow/util/bpacking_simd_internal.h | 72 +++ .../util/bpacking_simd_kernel_internal.h | 94 ++- cpp/src/arrow/util/bpacking_test.cc | 157 +++++ 12 files changed, 790 insertions(+), 291 deletions(-) diff --git a/cpp/src/arrow/util/bpacking.cc b/cpp/src/arrow/util/bpacking.cc index 1bf81df4f28f..4a2ec0f9c059 100644 --- a/cpp/src/arrow/util/bpacking.cc +++ b/cpp/src/arrow/util/bpacking.cc @@ -43,6 +43,23 @@ struct UnpackDynamicFunction { } }; +template +struct UnpackBiasDynamicFunction { + using FunctionType = decltype(&bpacking::unpack_bias_scalar); + + static constexpr auto targets() { + return std::array{ + ARROW_DISPATCH_TARGET_NONE(&bpacking::unpack_bias_scalar) // + ARROW_DISPATCH_TARGET_NEON(&bpacking::unpack_bias_neon) // + ARROW_DISPATCH_TARGET_SVE128(&bpacking::unpack_bias_sve128) // + ARROW_DISPATCH_TARGET_SVE256(&bpacking::unpack_bias_sve256) // + ARROW_DISPATCH_TARGET_SSE4_2(&bpacking::unpack_bias_sse4_2) // + ARROW_DISPATCH_TARGET_AVX2(&bpacking::unpack_bias_avx2) // + ARROW_DISPATCH_TARGET_AVX512(&bpacking::unpack_bias_avx512) // + }; + } +}; + } // namespace template @@ -57,4 +74,19 @@ template void unpack(const uint8_t*, uint16_t*, const UnpackOptions&); template void unpack(const uint8_t*, uint32_t*, const UnpackOptions&); template void unpack(const uint8_t*, uint64_t*, const UnpackOptions&); +template +void unpack_bias(const uint8_t* in, Uint* out, const UnpackOptions& opts, Uint bias) { + static const DynamicDispatch> dispatch; + return dispatch(in, out, opts, bias); +} + +template void unpack_bias(const uint8_t*, uint8_t*, const UnpackOptions&, + uint8_t); +template void unpack_bias(const uint8_t*, uint16_t*, const UnpackOptions&, + uint16_t); +template void unpack_bias(const uint8_t*, uint32_t*, const UnpackOptions&, + uint32_t); +template void unpack_bias(const uint8_t*, uint64_t*, const UnpackOptions&, + uint64_t); + } // namespace arrow::internal diff --git a/cpp/src/arrow/util/bpacking_dispatch_internal.h b/cpp/src/arrow/util/bpacking_dispatch_internal.h index 6ea6adee1800..e89ccc9e8083 100644 --- a/cpp/src/arrow/util/bpacking_dispatch_internal.h +++ b/cpp/src/arrow/util/bpacking_dispatch_internal.h @@ -31,23 +31,52 @@ namespace arrow::internal::bpacking { /// Unpack a zero bit packed array. -template -ARROW_FORCE_INLINE void unpack_null(const uint8_t* in, Uint* out, int batch_size) { - std::memset(out, 0, batch_size * sizeof(Uint)); +template +ARROW_FORCE_INLINE void unpack_null(const uint8_t* in, Uint* out, int batch_size, + Uint bias = Uint{}) { + if constexpr (kHasBias) { + // Every unpacked value is zero, so every output value is the bias. + std::fill(out, out + batch_size, bias); + } else { + std::memset(out, 0, batch_size * sizeof(Uint)); + } } /// Unpack a packed array where packed and unpacked values have exactly the same number of /// bits. -template -ARROW_FORCE_INLINE void unpack_full(const uint8_t* in, Uint* out, int batch_size) { +template +ARROW_FORCE_INLINE void unpack_full(const uint8_t* in, Uint* out, int batch_size, + Uint bias = Uint{}) { if constexpr (ARROW_LITTLE_ENDIAN == 1) { - std::memcpy(out, in, batch_size * sizeof(Uint)); + if constexpr (kHasBias) { + // Two details let this loop approach memcpy speed: + // 1. A constant-size memcpy for the load, not SafeLoadAs: SafeLoadAs + // builds an AlignedStorage per element and the vectorizer refuses it, + // while a fixed-size memcpy is just an unaligned load. + // 2. A restrict qualifier: `in` is a uint8_t*, so it may alias + // anything, including `out`. Without restating that they are + // distinct the compiler has to assume overlap and emits a scalar + // loop. + const uint8_t* ARROW_RESTRICT src = in; + Uint* ARROW_RESTRICT dst = out; + for (int k = 0; k < batch_size; k += 1) { + Uint val; + std::memcpy(&val, src + (k * sizeof(Uint)), sizeof(Uint)); + dst[k] = static_cast(val + bias); + } + } else { + std::memcpy(out, in, batch_size * sizeof(Uint)); + } } else { using bit_util::FromLittleEndian; using util::SafeLoadAs; for (int k = 0; k < batch_size; k += 1) { - out[k] = FromLittleEndian(SafeLoadAs(in + (k * sizeof(Uint)))); + Uint val = FromLittleEndian(SafeLoadAs(in + (k * sizeof(Uint)))); + if constexpr (kHasBias) { + val = static_cast(val + bias); + } + out[k] = val; } } } @@ -96,9 +125,9 @@ using SpreadBufferUint = std::conditional_t< /// This function works for all input batch sizes but is not the fastest. /// In prolog mode, instead of unpacking all required element, the function will /// stop if it finds a byte aligned value start. -template +template ARROW_FORCE_INLINE int unpack_exact(const uint8_t* in, const uint8_t* in_end, Uint* out, - int batch_size, int bit_offset) { + int batch_size, int bit_offset, Uint bias = Uint{}) { static_assert(kPackedBitWidth > 0); // For the epilog we adapt the max spread since better alignment give shorter spreads @@ -168,6 +197,9 @@ ARROW_FORCE_INLINE int unpack_exact(const uint8_t* in, const uint8_t* in_end, Ui } } + if constexpr (kHasBias) { + val = static_cast(val + bias); + } *out = val; out++; start_bit += kPackedBitWidth; @@ -190,12 +222,12 @@ ARROW_FORCE_INLINE int unpack_exact(const uint8_t* in, const uint8_t* in_end, Ui /// This is used to safely overread. /// Negative value to deduce from batch_size. template typename Unpacker, - typename UnpackedUInt> + bool kHasBias = false, typename UnpackedUInt> void unpack_width(const uint8_t* in, UnpackedUInt* out, int batch_size, int bit_offset, - int max_read_bytes) { + int max_read_bytes, UnpackedUInt bias = UnpackedUInt{}) { if constexpr (kPackedBitWidth == 0) { // Easy case to handle, simply setting memory to zero. - return unpack_null(in, out, batch_size); + return unpack_null(in, out, batch_size, bias); } else { // Number of bytes to read according to batch_size. const int bytes_batch = static_cast( @@ -206,8 +238,8 @@ void unpack_width(const uint8_t* in, UnpackedUInt* out, int batch_size, int bit_ const uint8_t* in_end = in + (max_read_bytes >= 0 ? max_read_bytes : bytes_batch); // In case of misalignment, we need to run the prolog until aligned. - int extracted = - unpack_exact(in, in_end, out, batch_size, bit_offset); + int extracted = unpack_exact( + in, in_end, out, batch_size, bit_offset, bias); // We either extracted everything or found a alignment const int start_bit = extracted * kPackedBitWidth + bit_offset; ARROW_DCHECK((extracted == batch_size) || ((start_bit) % 8 == 0)); @@ -218,7 +250,7 @@ void unpack_width(const uint8_t* in, UnpackedUInt* out, int batch_size, int bit_ if constexpr (kPackedBitWidth == 8 * sizeof(UnpackedUInt)) { // Only memcpy / static_cast - return unpack_full(in, out, batch_size); + return unpack_full(in, out, batch_size, bias); } else { using UnpackerForWidth = Unpacker; // Number of values extracted by one iteration of the kernel @@ -229,9 +261,29 @@ void unpack_width(const uint8_t* in, UnpackedUInt* out, int batch_size, int bit_ if constexpr (kValuesUnpacked > 0) { const uint8_t* in_last = in_end - kBytesRead; + // Whether this unpacker family folds the bias into its own stores. The + // xsimd kernels do; the generated scalar and AVX-512 families do not, so + // they get a second pass over the values the kernel just wrote. That + // pass is over kValuesUnpacked elements still in L1, not over the whole + // Some generated kernels do not accept a bias. Apply it in a fallback + // pass for correctness on those targets. + constexpr bool kUnpackerTakesBias = + requires(const uint8_t* i, UnpackedUInt* o, UnpackedUInt b) { + UnpackerForWidth::unpack(i, o, b); + }; // NOLINT(readability/braces) + // Running the optimized kernel for batch extraction while ((batch_size >= kValuesUnpacked) && (in <= in_last)) { - in = UnpackerForWidth::unpack(in, out); + if constexpr (kHasBias && kUnpackerTakesBias) { + in = UnpackerForWidth::unpack(in, out, bias); + } else { + in = UnpackerForWidth::unpack(in, out); + if constexpr (kHasBias) { + for (int k = 0; k < kValuesUnpacked; ++k) { + out[k] = static_cast(out[k] + bias); + } + } + } out += kValuesUnpacked; batch_size -= kValuesUnpacked; } @@ -245,406 +297,412 @@ void unpack_width(const uint8_t* in, UnpackedUInt* out, int batch_size, int bit_ // Running the epilog for the remaining values that don't fit in a kernel ARROW_DCHECK_GE(batch_size, 0); ARROW_COMPILER_ASSUME(batch_size >= 0); - unpack_exact(in, in_end, out, batch_size, - /* bit_offset= */ 0); + unpack_exact(in, in_end, out, batch_size, + /* bit_offset= */ 0, bias); } } } -template