Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
c1109a7
Fold frame bias into bit unpacking
sfc-gh-pgaur Oct 3, 2026
7770ac4
Add the PFOR codec and page format
sfc-gh-pgaur Oct 3, 2026
f24b140
Integrate PFOR with Parquet integer columns
sfc-gh-pgaur Oct 3, 2026
de76383
Test and benchmark Parquet PFOR integration
sfc-gh-pgaur Oct 3, 2026
706d549
Export PFOR templates from the Arrow library
sfc-gh-pgaur Oct 3, 2026
fe3d6ca
Document PFOR encoding selection
sfc-gh-pgaur Oct 3, 2026
c07a099
Add PFOR delta planning and searchable frames
sfc-gh-pgaur Sep 1, 2026
33feeff
Benchmark PFOR delta mode on correlated columns
sfc-gh-pgaur Sep 1, 2026
4c02050
Reject unprofitable PFOR delta plans early
sfc-gh-pgaur Sep 3, 2026
4fd7ac7
Gate PFOR and delta mode with writer properties
sfc-gh-pgaur Sep 3, 2026
25d786a
Validate PFOR delta metadata and page bounds
sfc-gh-pgaur Sep 3, 2026
6e15dc9
Let benchmarks force delta vectors
sfc-gh-pgaur Sep 28, 2026
abbf366
Add lane-interleaved packing to PFOR
sfc-gh-pgaur Sep 4, 2026
3c2054c
Integrate and test interleaved PFOR writing
sfc-gh-pgaur Sep 4, 2026
b69d4f0
Benchmark sequential and interleaved PFOR layouts
sfc-gh-pgaur Sep 4, 2026
ddb3faa
Add FastLanes-style lane-parallel delta kernels
sfc-gh-pgaur Sep 5, 2026
02564e2
Decode lane-parallel delta pages
sfc-gh-pgaur Sep 8, 2026
c7f3dd0
Frame FastLanes delta pages
sfc-gh-pgaur Sep 9, 2026
954f10a
Benchmark interleaved PFOR on the corpus
sfc-gh-pgaur Sep 12, 2026
94a5b81
Cap bit-unpack dispatch at AVX2
sfc-gh-pgaur Sep 13, 2026
e9e877f
Fuse FL_ORDER transpose into unpacking
sfc-gh-pgaur Sep 13, 2026
5d0f349
Dispatch interleaved kernels by ISA
sfc-gh-pgaur Sep 16, 2026
e0c5250
Share one column corpus across PFOR benchmarks
sfc-gh-pgaur Sep 19, 2026
7f697e0
Transpose FL_ORDER blocks in NEON registers
sfc-gh-pgaur Sep 20, 2026
e27369f
Generalize block kernels over element width
sfc-gh-pgaur Oct 4, 2026
36f0854
Add a bit-unpack width and footprint benchmark
sfc-gh-pgaur Oct 5, 2026
37104b8
Shift narrow lanes in place on AVX2
sfc-gh-pgaur Oct 5, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -94,3 +94,6 @@ rat.txt

# for ODBC DLL
*.rc

# Local out-of-tree benchmark build dir
cpp/build-bench/
25 changes: 25 additions & 0 deletions cpp/src/arrow/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -551,6 +551,7 @@ set(ARROW_UTIL_SRCS
util/decimal.cc
util/delimiting.cc
util/dict_util.cc
util/fastlanes/interleaved_pfor_baseline.cc
util/fixed_width_internal.cc
util/float16.cc
util/formatting.cc
Expand All @@ -566,6 +567,8 @@ set(ARROW_UTIL_SRCS
util/math_internal.cc
util/memory.cc
util/mutex.cc
util/pfor/pfor.cc
util/pfor/pfor_wrapper.cc
util/ree_util.cc
util/secure_string.cc
util/string.cc
Expand All @@ -591,6 +594,28 @@ append_runtime_avx512_src(ARROW_UTIL_SRCS util/bpacking_simd_avx512.cc)
append_runtime_sve128_src(ARROW_UTIL_SRCS util/bpacking_simd_128_alt.cc)
append_runtime_sve256_src(ARROW_UTIL_SRCS util/bpacking_simd_256.cc)

# One source compiled once per instruction set, the way bpacking_simd_256.cc is
# registered for both AVX2 and SVE256 above: the file forks on the platform
# macros, and only one of them is ever defined on a given target.
append_runtime_avx2_src(ARROW_UTIL_SRCS util/fastlanes/interleaved_pfor_simd.cc)
append_runtime_sve128_src(ARROW_UTIL_SRCS util/fastlanes/interleaved_pfor_simd.cc)

# The interleaved kernels are portable C++ with no intrinsics, so what they
# compile to is decided by the optimizer, not by the source. At width 16 gcc 11.5
# emits 289 instructions and no vector operations at -O2, 322 with 81 vector
# operations at Release's -O2 -ftree-vectorize, and 849 with 513 at -O3; the
# published throughput figures for this layout were all taken at the last of
# those. APPEND rather than a plain set, because the two calls above have already
# put the instruction-set flags in this property. Release only -- Debug and the
# sanitizer builds want their own level, and MSVC spells this differently.
if(NOT MSVC)
set_property(SOURCE util/fastlanes/interleaved_pfor_baseline.cc
util/fastlanes/interleaved_pfor_simd.cc
APPEND
PROPERTY COMPILE_OPTIONS
"$<$<OR:$<CONFIG:Release>,$<CONFIG:RelWithDebInfo>>:-O3>")
endif()

if(ARROW_WITH_BROTLI)
list(APPEND ARROW_UTIL_SRCS util/compression_brotli.cc)
endif()
Expand Down
2 changes: 2 additions & 0 deletions cpp/src/arrow/meson.build
Original file line number Diff line number Diff line change
Expand Up @@ -204,6 +204,8 @@ arrow_util_srcs = [
'util/math_internal.cc',
'util/memory.cc',
'util/mutex.cc',
'util/pfor/pfor.cc',
'util/pfor/pfor_wrapper.cc',
'util/ree_util.cc',
'util/secure_string.cc',
'util/string.cc',
Expand Down
10 changes: 10 additions & 0 deletions cpp/src/arrow/util/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -118,6 +118,16 @@ add_arrow_test(threading-utility-test
test_common.cc
thread_pool_test.cc)

add_arrow_test(fastlanes-kernels-test
SOURCES fastlanes/fastlanes_kernels_test.cc)

add_arrow_test(pfor-test SOURCES pfor/pfor_test.cc)

add_arrow_benchmark(fastlanes/fastlanes_kernels_benchmark EXTRA_LINK_LIBS
${ARROW_XSIMD})

add_arrow_benchmark(pfor/pfor_benchmark)

add_arrow_benchmark(bit_block_counter_benchmark)
add_arrow_benchmark(bit_util_benchmark)
add_arrow_benchmark(bitmap_reader_benchmark)
Expand Down
39 changes: 38 additions & 1 deletion cpp/src/arrow/util/bpacking.cc
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,29 @@ struct UnpackDynamicFunction {
ARROW_DISPATCH_TARGET_SVE256(&bpacking::unpack_sve256<Uint>) //
ARROW_DISPATCH_TARGET_SSE4_2(&bpacking::unpack_sse4_2<Uint>) //
ARROW_DISPATCH_TARGET_AVX2(&bpacking::unpack_avx2<Uint>) //
ARROW_DISPATCH_TARGET_AVX512(&bpacking::unpack_avx512<Uint>) //
// Cap bit-unpack dispatch at 256 bits. The generated AVX-512 kernels
// assemble vectors from scalar loads and use out-of-line calls, making
// them slower than the AVX2 path. Re-enable this target when its kernels
// use vector loads directly.
// ARROW_DISPATCH_TARGET_AVX512(&bpacking::unpack_avx512<Uint>) //
};
}
};

template <typename Uint>
struct UnpackBiasDynamicFunction {
using FunctionType = decltype(&bpacking::unpack_bias_scalar<Uint>);

static constexpr auto targets() {
return std::array{
ARROW_DISPATCH_TARGET_NONE(&bpacking::unpack_bias_scalar<Uint>) //
ARROW_DISPATCH_TARGET_NEON(&bpacking::unpack_bias_neon<Uint>) //
ARROW_DISPATCH_TARGET_SVE128(&bpacking::unpack_bias_sve128<Uint>) //
ARROW_DISPATCH_TARGET_SVE256(&bpacking::unpack_bias_sve256<Uint>) //
ARROW_DISPATCH_TARGET_SSE4_2(&bpacking::unpack_bias_sse4_2<Uint>) //
ARROW_DISPATCH_TARGET_AVX2(&bpacking::unpack_bias_avx2<Uint>) //
// Capped at 256 bits for the reason given in UnpackDynamicFunction above.
// ARROW_DISPATCH_TARGET_AVX512(&bpacking::unpack_bias_avx512<Uint>) //
};
}
};
Expand All @@ -57,4 +79,19 @@ template void unpack<uint16_t>(const uint8_t*, uint16_t*, const UnpackOptions&);
template void unpack<uint32_t>(const uint8_t*, uint32_t*, const UnpackOptions&);
template void unpack<uint64_t>(const uint8_t*, uint64_t*, const UnpackOptions&);

template <typename Uint>
void unpack_bias(const uint8_t* in, Uint* out, const UnpackOptions& opts, Uint bias) {
static const DynamicDispatch<UnpackBiasDynamicFunction<Uint>> dispatch;
return dispatch(in, out, opts, bias);
}

template void unpack_bias<uint8_t>(const uint8_t*, uint8_t*, const UnpackOptions&,
uint8_t);
template void unpack_bias<uint16_t>(const uint8_t*, uint16_t*, const UnpackOptions&,
uint16_t);
template void unpack_bias<uint32_t>(const uint8_t*, uint32_t*, const UnpackOptions&,
uint32_t);
template void unpack_bias<uint64_t>(const uint8_t*, uint64_t*, const UnpackOptions&,
uint64_t);

} // namespace arrow::internal
Loading
Loading