Skip to content

Commit ee608e6

Browse files
committed
[C++][Parquet] Scan DELTA_BINARY_PACKED deltas a vector at a time
The value-at-a-time prefix sum becomes a helper that scans whole registers with a log-step inclusive scan and finishes the remainder one value at a time. Adding the frame of reference before the scan turns its running multiple into a term the scan produces rather than a multiply per lane, and the running total is carried between registers in a vector register rather than through a general-purpose one. The xsimd include and the vector loop are guarded on ARROW_HAVE_NEON or ARROW_HAVE_SSE4_2, so a build with neither compiles the scalar loop alone. The vector loop is compiled only where a register holds four or more values, which at the 128-bit baseline vectorizes 32-bit values and leaves 64-bit ones on the value-at-a-time loop. Also adds a decode benchmark on non-decreasing values, the shape this encoding is usually chosen for. Arithmetic is unchanged: every term stays in the unsigned type, so the wrapping the format specifies is preserved and no decoded value changes.
1 parent bd877ee commit ee608e6

3 files changed

Lines changed: 152 additions & 11 deletions

File tree

‎cpp/src/parquet/decoder.cc‎

Lines changed: 72 additions & 11 deletions
Original file line numberDiff line numberDiff line change
@@ -32,6 +32,10 @@
3232
#include <utility>
3333
#include <vector>
3434

35+
#if defined(ARROW_HAVE_NEON) || defined(ARROW_HAVE_SSE4_2)
36+
# include <xsimd/xsimd.hpp>
37+
#endif
38+
3539
#include "arrow/array.h"
3640
#include "arrow/array/builder_binary.h"
3741
#include "arrow/array/builder_dict.h"
@@ -1437,6 +1441,71 @@ class DictByteArrayDecoderImpl : public DictDecoderImpl<ByteArrayType> {
14371441
// ----------------------------------------------------------------------
14381442
// DELTA_BINARY_PACKED decoder
14391443

1444+
namespace {
1445+
1446+
#if defined(ARROW_HAVE_NEON) || defined(ARROW_HAVE_SSE4_2)
1447+
// One step of an inclusive scan, recursed at compile time over powers of two. Each step
1448+
// adds the vector to a copy of itself slid up by kShift lanes, zero-filling the lanes
1449+
// it vacates, so after the last step lane k holds the sum of lanes 0 through k.
1450+
template <std::size_t kShift, typename Batch>
1451+
Batch InclusiveScanSteps(Batch v) {
1452+
if constexpr (kShift < Batch::size) {
1453+
v += xsimd::slide_left<kShift * sizeof(typename Batch::value_type)>(v);
1454+
return InclusiveScanSteps<kShift * 2>(v);
1455+
} else {
1456+
return v;
1457+
}
1458+
}
1459+
#endif
1460+
1461+
// Turns a run of deltas into the values they encode, in place: on return element k holds
1462+
// `last + (k + 1) * min_delta + sum of deltas 0..k`, and the return value is the last
1463+
// element. Every term is unsigned, so the wrapping matches the loop this replaces.
1464+
template <typename T>
1465+
std::make_unsigned_t<T> PrefixSumDeltas(T* values, int num_values,
1466+
std::make_unsigned_t<T> min_delta,
1467+
std::make_unsigned_t<T> last) {
1468+
using UT = std::make_unsigned_t<T>;
1469+
int i = 0;
1470+
1471+
#if defined(ARROW_HAVE_NEON) || defined(ARROW_HAVE_SSE4_2)
1472+
using Batch = xsimd::batch<UT>;
1473+
constexpr int kLanes = static_cast<int>(Batch::size);
1474+
// At two lanes the scan loses to the additions it replaces, so it is compiled only
1475+
// where a register holds four or more values; narrower ones use the loop below.
1476+
if constexpr (kLanes >= 4) {
1477+
// Broadcast pattern for the last lane, which carries the running value into the
1478+
// next vector without a round trip through a general-purpose register.
1479+
struct LastLane {
1480+
static constexpr unsigned get(unsigned /*index*/, unsigned size) {
1481+
return size - 1;
1482+
}
1483+
};
1484+
const auto last_lane =
1485+
xsimd::make_batch_constant<UT, LastLane, xsimd::default_arch>();
1486+
const Batch min_delta_v(min_delta);
1487+
Batch carry(last);
1488+
for (; i + kLanes <= num_values; i += kLanes) {
1489+
// Adding the frame before the scan turns its running multiple into a term the
1490+
// scan produces, rather than a multiply per lane.
1491+
Batch v = xsimd::bitwise_cast<UT>(xsimd::batch<T>::load_unaligned(values + i));
1492+
v = InclusiveScanSteps<1>(v + min_delta_v) + carry;
1493+
xsimd::bitwise_cast<T>(v).store_unaligned(values + i);
1494+
carry = xsimd::swizzle(v, last_lane);
1495+
}
1496+
last = carry.get(0);
1497+
}
1498+
#endif
1499+
1500+
for (; i < num_values; ++i) {
1501+
last += min_delta + static_cast<UT>(values[i]);
1502+
values[i] = static_cast<T>(last);
1503+
}
1504+
return last;
1505+
}
1506+
1507+
} // namespace
1508+
14401509
template <typename DType>
14411510
class DeltaBitPackDecoder : public TypedDecoderImpl<DType> {
14421511
public:
@@ -1694,17 +1763,9 @@ class DeltaBitPackDecoder : public TypedDecoderImpl<DType> {
16941763
values_decode) {
16951764
ParquetException::EofException();
16961765
}
1697-
// Held in locals because a store to `buffer` may alias these members, which
1698-
// would force a reload of each per value.
1699-
UT last = static_cast<UT>(last_value_);
1700-
const UT min_delta = static_cast<UT>(min_delta_);
1701-
for (int j = 0; j < values_decode; ++j) {
1702-
// Addition between min_delta, packed int and last_value should be treated as
1703-
// unsigned addition. Overflow is as expected.
1704-
last += min_delta + static_cast<UT>(buffer[i + j]);
1705-
buffer[i + j] = last;
1706-
}
1707-
last_value_ = static_cast<T>(last);
1766+
last_value_ = static_cast<T>(PrefixSumDeltas(buffer + i, values_decode,
1767+
static_cast<UT>(min_delta_),
1768+
static_cast<UT>(last_value_)));
17081769
}
17091770
// The last miniblock of a coalesced run becomes the current one, fully drained.
17101771
mini_block_idx_ += mini_blocks_coalesced;

‎cpp/src/parquet/encoding_benchmark.cc‎

Lines changed: 40 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -675,6 +675,22 @@ static auto MakeDeltaBitPackingInputNarrow(size_t length) {
675675
return numbers;
676676
}
677677

678+
// Non-decreasing values, the shape this encoding is usually chosen for. The deltas come
679+
// from the same 1000-wide range as Narrow, so the two inputs differ in the order of the
680+
// values and not in the width a delta is packed into.
681+
template <typename DType>
682+
static auto MakeDeltaBitPackingInputNarrowSorted(size_t length) {
683+
using T = typename DType::c_type;
684+
auto numbers = std::vector<T>(length);
685+
::arrow::randint<T, T>(length, 0, 1000, &numbers);
686+
T value = 0;
687+
for (auto& number : numbers) {
688+
value = static_cast<T>(value + number);
689+
number = value;
690+
}
691+
return numbers;
692+
}
693+
678694
template <typename DType>
679695
static auto MakeDeltaBitPackingInputWide(size_t length) {
680696
using T = typename DType::c_type;
@@ -713,6 +729,16 @@ static void BM_DeltaBitPackingEncode_Int64_Narrow(benchmark::State& state) {
713729
BM_DeltaBitPackingEncode<Int64Type>(state, MakeDeltaBitPackingInputNarrow<Int64Type>);
714730
}
715731

732+
static void BM_DeltaBitPackingEncode_Int32_NarrowSorted(benchmark::State& state) {
733+
BM_DeltaBitPackingEncode<Int32Type>(state,
734+
MakeDeltaBitPackingInputNarrowSorted<Int32Type>);
735+
}
736+
737+
static void BM_DeltaBitPackingEncode_Int64_NarrowSorted(benchmark::State& state) {
738+
BM_DeltaBitPackingEncode<Int64Type>(state,
739+
MakeDeltaBitPackingInputNarrowSorted<Int64Type>);
740+
}
741+
716742
static void BM_DeltaBitPackingEncode_Int32_Wide(benchmark::State& state) {
717743
BM_DeltaBitPackingEncode<Int32Type>(state, MakeDeltaBitPackingInputWide<Int32Type>);
718744
}
@@ -725,6 +751,8 @@ BENCHMARK(BM_DeltaBitPackingEncode_Int32_Fixed)->Range(MIN_RANGE, MAX_RANGE);
725751
BENCHMARK(BM_DeltaBitPackingEncode_Int64_Fixed)->Range(MIN_RANGE, MAX_RANGE);
726752
BENCHMARK(BM_DeltaBitPackingEncode_Int32_Narrow)->Range(MIN_RANGE, MAX_RANGE);
727753
BENCHMARK(BM_DeltaBitPackingEncode_Int64_Narrow)->Range(MIN_RANGE, MAX_RANGE);
754+
BENCHMARK(BM_DeltaBitPackingEncode_Int32_NarrowSorted)->Range(MIN_RANGE, MAX_RANGE);
755+
BENCHMARK(BM_DeltaBitPackingEncode_Int64_NarrowSorted)->Range(MIN_RANGE, MAX_RANGE);
728756
BENCHMARK(BM_DeltaBitPackingEncode_Int32_Wide)->Range(MIN_RANGE, MAX_RANGE);
729757
BENCHMARK(BM_DeltaBitPackingEncode_Int64_Wide)->Range(MIN_RANGE, MAX_RANGE);
730758

@@ -762,6 +790,16 @@ static void BM_DeltaBitPackingDecode_Int64_Narrow(benchmark::State& state) {
762790
BM_DeltaBitPackingDecode<Int64Type>(state, MakeDeltaBitPackingInputNarrow<Int64Type>);
763791
}
764792

793+
static void BM_DeltaBitPackingDecode_Int32_NarrowSorted(benchmark::State& state) {
794+
BM_DeltaBitPackingDecode<Int32Type>(state,
795+
MakeDeltaBitPackingInputNarrowSorted<Int32Type>);
796+
}
797+
798+
static void BM_DeltaBitPackingDecode_Int64_NarrowSorted(benchmark::State& state) {
799+
BM_DeltaBitPackingDecode<Int64Type>(state,
800+
MakeDeltaBitPackingInputNarrowSorted<Int64Type>);
801+
}
802+
765803
static void BM_DeltaBitPackingDecode_Int32_Wide(benchmark::State& state) {
766804
BM_DeltaBitPackingDecode<Int32Type>(state, MakeDeltaBitPackingInputWide<Int32Type>);
767805
}
@@ -774,6 +812,8 @@ BENCHMARK(BM_DeltaBitPackingDecode_Int32_Fixed)->Range(MIN_RANGE, MAX_RANGE);
774812
BENCHMARK(BM_DeltaBitPackingDecode_Int64_Fixed)->Range(MIN_RANGE, MAX_RANGE);
775813
BENCHMARK(BM_DeltaBitPackingDecode_Int32_Narrow)->Range(MIN_RANGE, MAX_RANGE);
776814
BENCHMARK(BM_DeltaBitPackingDecode_Int64_Narrow)->Range(MIN_RANGE, MAX_RANGE);
815+
BENCHMARK(BM_DeltaBitPackingDecode_Int32_NarrowSorted)->Range(MIN_RANGE, MAX_RANGE);
816+
BENCHMARK(BM_DeltaBitPackingDecode_Int64_NarrowSorted)->Range(MIN_RANGE, MAX_RANGE);
777817
BENCHMARK(BM_DeltaBitPackingDecode_Int32_Wide)->Range(MIN_RANGE, MAX_RANGE);
778818
BENCHMARK(BM_DeltaBitPackingDecode_Int64_Wide)->Range(MIN_RANGE, MAX_RANGE);
779819

‎cpp/src/parquet/encoding_test.cc‎

Lines changed: 40 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -24,6 +24,7 @@
2424
#include <limits>
2525
#include <numeric>
2626
#include <span>
27+
#include <type_traits>
2728
#include <utility>
2829
#include <vector>
2930

@@ -2188,6 +2189,45 @@ TYPED_TEST(TestDeltaBitPackEncoding, MiniblockBitWidthRuns) {
21882189
}
21892190
}
21902191

2192+
TYPED_TEST(TestDeltaBitPackEncoding, PrefixSumVectorAndTail) {
2193+
// A decoder may accumulate the running total several deltas at a time and finish the
2194+
// remainder one at a time. Walk every residual bit width at enough lengths to leave
2195+
// every remainder such a group can leave, so each width is decoded through the
2196+
// grouped path, through the remainder, and across the hand-off between them. A
2197+
// non-zero frame checks the running multiple, and wrapping the total checks that no
2198+
// term is signed.
2199+
using T = typename TypeParam::c_type;
2200+
using UT = std::make_unsigned_t<T>;
2201+
constexpr int kBits = static_cast<int>(sizeof(T) * 8);
2202+
2203+
auto make_values = [](int width, T frame, int num_deltas) {
2204+
std::vector<T> values;
2205+
values.reserve(num_deltas + 1);
2206+
const UT spread = width == kBits ? ~UT{0} : static_cast<UT>((UT{1} << width) - 1);
2207+
// Two deltas in three sit at the frame, so it is the smallest in every miniblock.
2208+
UT current = 0;
2209+
values.push_back(static_cast<T>(current));
2210+
for (int i = 0; i < num_deltas; ++i) {
2211+
current = static_cast<UT>(current + static_cast<UT>(frame) +
2212+
(i % 3 == 0 ? spread : UT{0}));
2213+
values.push_back(static_cast<T>(current));
2214+
}
2215+
return values;
2216+
};
2217+
2218+
for (int width = 0; width <= kBits; ++width) {
2219+
for (const T frame : {T{0}, static_cast<T>(-5), T{7}}) {
2220+
// 16-23 leaves every remainder for group sizes up to eight; 201 crosses a block
2221+
// boundary with a remainder left.
2222+
for (const int num_deltas : {16, 17, 18, 19, 20, 21, 22, 23, 201}) {
2223+
ARROW_SCOPED_TRACE("width = ", width, ", frame = ", static_cast<int64_t>(frame),
2224+
", num_deltas = ", num_deltas);
2225+
this->CheckRoundtripWithValues(make_values(width, frame, num_deltas));
2226+
}
2227+
}
2228+
}
2229+
}
2230+
21912231
// ----------------------------------------------------------------------
21922232
// PFOR encode/decode tests.
21932233

0 commit comments

Comments
 (0)