Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
106 commits
Select commit Hold shift + click to select a range
7cc6bf5
Add PFOR core algorithm and tests
sfc-gh-pgaur Apr 20, 2026
378b3ee
Integrate PFOR encoding into parquet encoder/decoder
sfc-gh-pgaur Apr 21, 2026
6a1d6fb
Add PFOR encoding benchmark
sfc-gh-pgaur Apr 21, 2026
003818c
Use signed integer types consistently per Arrow style guide
sfc-gh-pgaur Jun 3, 2026
f2459cd
Take buffer parameters as spans in the PFOR wire routines
sfc-gh-pgaur Jun 3, 2026
df91383
Return Result<T>/Status on decode paths instead of ARROW_DCHECK
sfc-gh-pgaur Jun 3, 2026
5950438
Assert little-endian and replace reinterpret_cast with SafeCopy
sfc-gh-pgaur Jun 3, 2026
7dfaa76
Unpack a PFOR vector in one call instead of a batch loop
sfc-gh-pgaur Jun 3, 2026
ad3dc1e
Add pragma GCC unroll/ivdep to decode loops for better vectorization
sfc-gh-pgaur Jun 3, 2026
4699aab
Add PforEncodedVectorView for zero-copy decode path
sfc-gh-pgaur Jun 3, 2026
8976c6e
Let the encode path choose its vector size
sfc-gh-pgaur Jun 3, 2026
ab7852e
Fix pfor_test.cc: unwrap Result<> from PforVectorInfo::Load()
sfc-gh-pgaur Jun 3, 2026
cf9e94f
Harden PFOR page header loading
sfc-gh-pgaur Jun 14, 2026
a229dcf
Align PFOR interfaces with Arrow buffer types
sfc-gh-pgaur Jun 15, 2026
5017719
Encapsulate and validate PFOR vector metadata
sfc-gh-pgaur Jun 15, 2026
fa1d6f3
Build the PFOR comparison benchmark
sfc-gh-pgaur Jun 25, 2026
9e84c83
Use bit_util helpers in PFOR and note the incremental paths
sfc-gh-pgaur Jun 26, 2026
f711c4e
Drop Snowflake attribution from PFOR header comments
sfc-gh-pgaur Jun 26, 2026
2f3f77c
Fix IsPowerOf2(int32_t) ambiguity in pfor_wrapper.cc
sfc-gh-pgaur Jun 28, 2026
fe45511
Add portable FastLanes bit packing and benchmarks
sfc-gh-pgaur Jun 28, 2026
0e0bb77
Add a flat-output decode for FastLanes-FOR and benchmark it
sfc-gh-pgaur Jun 28, 2026
98df87a
Add the inverse FL_ORDER index mapping
sfc-gh-pgaur Jun 28, 2026
5fdba12
Add FastLanes packing and output modes to PFOR
sfc-gh-pgaur Jun 28, 2026
3517a90
Add ARROW_RELEASE_O3 to keep CMake's default Release -O3
sfc-gh-pgaur Jun 28, 2026
72b1d6c
Cover bit_width 0 and 32 in the transposed-output tests
sfc-gh-pgaur Jul 20, 2026
36b0723
Adapt PFOR to current Arrow utility APIs
sfc-gh-pgaur Jul 20, 2026
ef93df3
Add a packing mode that interleaves without FL_ORDER
sfc-gh-pgaur Jul 20, 2026
84554fe
Vectorize BitPack decode without heap scratch
sfc-gh-pgaur Jul 24, 2026
30a7f86
Speed up PFOR encode with a wider histogram and stack deltas
sfc-gh-pgaur Jul 24, 2026
ad3c093
Skip FOR-add pass in PFOR decode when frame-of-reference is 0
sfc-gh-pgaur Jul 24, 2026
b0bf9ea
Expand the PFOR benchmark corpus
sfc-gh-pgaur Jul 24, 2026
9b645db
Remove experimental FastLanes modes from PFOR
sfc-gh-pgaur Jul 24, 2026
e87e837
Benchmark PFOR on int64 columns
sfc-gh-pgaur Jul 25, 2026
3cdfc0a
Fold frame bias into the bit unpacker
sfc-gh-pgaur Aug 21, 2026
4639699
Store the PFOR bit width in 7 bits, not 6
sfc-gh-pgaur Aug 21, 2026
cfa2a7a
Apply frame bias in the PFOR unpacker
sfc-gh-pgaur Aug 21, 2026
64fb483
Fix PFOR check-in warnings
sfc-gh-pgaur Aug 25, 2026
b9dc3ab
List PFOR in SupportedEncodings for INT32 and INT64
sfc-gh-pgaur Aug 25, 2026
fcdb5d9
Use std::bit_width instead of __builtin_clz in PFOR
sfc-gh-pgaur Aug 25, 2026
04f7e35
Apply clang-format 18 to the PFOR sources
sfc-gh-pgaur Aug 25, 2026
3a8c730
Reject a PFOR page header that disagrees with its buffer
sfc-gh-pgaur Aug 25, 2026
b203d7a
Count PFOR exceptions in an unsigned field
sfc-gh-pgaur Aug 25, 2026
3c7b45b
Put the output buffer last in PforWrapper::Decode
sfc-gh-pgaur Aug 25, 2026
890b2e3
Return Status from PforWrapper::Encode
sfc-gh-pgaur Aug 25, 2026
341ada9
Decode PFOR pages with null slots through the Arrow path
sfc-gh-pgaur Aug 25, 2026
5752bc2
Note that incremental PFOR encode and decode come later
sfc-gh-pgaur Aug 25, 2026
dd76da9
Correct the bit_width mask comments in PforVectorInfo
sfc-gh-pgaur Aug 26, 2026
6eae0fc
Harden PFOR decoding state and metadata
sfc-gh-pgaur Aug 26, 2026
ae01271
Build PFOR into libarrow instead of recompiling it per target
sfc-gh-pgaur Aug 27, 2026
b7bf9e0
Stop PforDecoder from shadowing its base class page state
sfc-gh-pgaur Aug 27, 2026
51eb634
Simplify PFOR layout metadata and serialization
sfc-gh-pgaur Aug 27, 2026
d734488
Note the missing PLAIN fallback in PforEncoder
sfc-gh-pgaur Aug 27, 2026
fc06a80
Type-parameterize the PFOR tests over int32 and int64
sfc-gh-pgaur Aug 27, 2026
0cb7e30
Take the PFOR value count from the page's own header
sfc-gh-pgaur Sep 3, 2026
f8b807c
Validate the PFOR element count and offset chain before decoding
sfc-gh-pgaur Sep 3, 2026
fa49c94
Encode an all-null PFOR page as a bare header
sfc-gh-pgaur Sep 3, 2026
20c7341
Serialize PFOR little-endian instead of refusing to build
sfc-gh-pgaur Sep 3, 2026
0d65628
Build the PFOR sources under meson too
sfc-gh-pgaur Sep 3, 2026
1cb5d69
Document the cost of batched PFOR reads
sfc-gh-pgaur Sep 3, 2026
606068c
Add end-to-end PFOR tests through the Arrow reader and writer
sfc-gh-pgaur Sep 3, 2026
d8689bd
Name the PFOR headers _internal.h
sfc-gh-pgaur Sep 3, 2026
4d65820
Bound the sequential unpacker at the page, not at the vector
sfc-gh-pgaur Sep 5, 2026
33c976b
Use Arrow's section-rule style for the PFOR comments
sfc-gh-pgaur Sep 19, 2026
1298e2b
Decode PFOR pages one vector at a time
sfc-gh-pgaur Sep 28, 2026
e0ea1e8
Give PFOR a delta mode and a searchable frame
sfc-gh-pgaur Sep 1, 2026
6ff07a9
Reduce the cost of PFOR frame selection
sfc-gh-pgaur Sep 1, 2026
e522f9b
Benchmark PFOR delta mode on correlated columns
sfc-gh-pgaur Sep 1, 2026
6fa1b2c
Reject unprofitable PFOR delta plans from a sample
sfc-gh-pgaur Sep 3, 2026
874ccd5
Gate PFOR and delta mode with writer properties
sfc-gh-pgaur Sep 3, 2026
acac1bf
Validate PFOR delta metadata before decoding
sfc-gh-pgaur Sep 3, 2026
d4f4524
Note that the delta prefix sum could fold into the unpack kernel
sfc-gh-pgaur Sep 3, 2026
5bcd09b
Measure the unpacker's page bound from the header this vector wrote
sfc-gh-pgaur Sep 5, 2026
46323ed
Clarify delta-mode comments and test terminology
sfc-gh-pgaur Sep 19, 2026
70fb031
Let a caller require the delta representation
sfc-gh-pgaur Sep 28, 2026
e4f76c4
[C++][Parquet] Coalesce equal-width DELTA_BINARY_PACKED miniblocks
sfc-gh-pgaur Sep 18, 2026
28cff0b
[C++][Parquet] Scan DELTA_BINARY_PACKED deltas a vector at a time
sfc-gh-pgaur Sep 18, 2026
d153ad7
Add lane-interleaved bit packing for 32-bit blocks
sfc-gh-pgaur Sep 4, 2026
fc434fa
Add a lane-interleaved packing mode to PFOR pages
sfc-gh-pgaur Sep 4, 2026
f96e978
Test the interleaved bit-packing layout
sfc-gh-pgaur Sep 4, 2026
688ed7d
Let a writer ask PFOR for the interleaved bit-packing layout
sfc-gh-pgaur Sep 4, 2026
00acb86
Test the interleaved layout through a written Parquet file
sfc-gh-pgaur Sep 4, 2026
eaa7967
Benchmark the interleaved layout against the sequential one
sfc-gh-pgaur Sep 4, 2026
e2f300b
Benchmark a page-sized destination, not only powers of 1024
sfc-gh-pgaur Sep 5, 2026
7128754
Clarify the scope of interleaved PFOR measurements
sfc-gh-pgaur Sep 5, 2026
b05950b
Measure the FastLanes paper's lane assignment for DELTA
sfc-gh-pgaur Sep 5, 2026
90b0664
Fuse FL_ORDER transpose with the delta prefix sum
sfc-gh-pgaur Sep 5, 2026
12457a9
Read the lane-parallel delta layout through a Parquet decoder
sfc-gh-pgaur Sep 8, 2026
7fb14d7
Benchmark lane-delta decoding against Arrow's decoder
sfc-gh-pgaur Sep 8, 2026
c9f67f9
Refuse a lane-delta page that does not describe itself
sfc-gh-pgaur Sep 8, 2026
b9bb0b9
Stop the entry-point reader at the end of its stream
sfc-gh-pgaur Sep 9, 2026
9f92ae8
Frame the paper's lane assignment as a page
sfc-gh-pgaur Sep 9, 2026
751b502
Add a page format for lane-parallel deltas
sfc-gh-pgaur Sep 9, 2026
a23d2be
Gate the interleaved-layout tests behind the PFOR preview flag
sfc-gh-pgaur Sep 10, 2026
f4e1c2e
Benchmark interleaved PFOR decoding on the column corpus
sfc-gh-pgaur Sep 12, 2026
c540245
Hold delta selection fixed in PFOR layout benchmarks
sfc-gh-pgaur Sep 12, 2026
0c91f8d
Cap bit-unpack SIMD dispatch at 256 bits
sfc-gh-pgaur Sep 13, 2026
d8f503a
Fuse the FL_ORDER transpose into unpacking
sfc-gh-pgaur Sep 13, 2026
822a0f8
Measure layout decoders across working-set sizes
sfc-gh-pgaur Sep 14, 2026
b790aa7
Document valid layout benchmark comparisons
sfc-gh-pgaur Sep 14, 2026
de1fd1e
Dispatch interleaved PFOR kernels by instruction set
sfc-gh-pgaur Sep 16, 2026
4ba439f
Align interleaved layout comments and names with Arrow style
sfc-gh-pgaur Sep 19, 2026
3a98300
Transpose FL_ORDER blocks in NEON registers
sfc-gh-pgaur Sep 20, 2026
728f75c
Keep PFOR encode options aggregate-initializable
sfc-gh-pgaur Oct 2, 2026
ec9ad5a
Skip the compressor benchmarks a build did not enable
sfc-gh-pgaur Oct 2, 2026
14aca7c
Propagate the PFOR packing mode through vector reads
sfc-gh-pgaur Oct 2, 2026
199407c
Use standard terminology for transpose operations
sfc-gh-pgaur Oct 2, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -94,3 +94,6 @@ rat.txt

# for ODBC DLL
*.rc

# Local out-of-tree benchmark build dir
cpp/build-bench/
3 changes: 3 additions & 0 deletions cpp/cmake_modules/DefineOptions.cmake
Original file line number Diff line number Diff line change
Expand Up @@ -194,6 +194,9 @@ takes precedence over ccache if a storage backend is configured" ON)

define_option(ARROW_GGDB_DEBUG "Pass -ggdb flag to debug builds" ON)

define_option(ARROW_RELEASE_O3
"Keep CMake's default -O3 in Release builds instead of -O2" OFF)

define_option(ARROW_WITH_MUSL "Whether the system libc is musl or not" OFF)

define_option(ARROW_ENABLE_THREADING "Enable threading in Arrow core" ON)
Expand Down
13 changes: 9 additions & 4 deletions cpp/cmake_modules/SetupCxxFlags.cmake
Original file line number Diff line number Diff line change
Expand Up @@ -632,20 +632,25 @@ endif()
# Same as Release, except with debug symbols enabled.

if(NOT MSVC)
# CMake's default Release flags are "-O3 -DNDEBUG"; the appends below
# downgrade them to -O2, since the last -O flag wins in GCC. Set
# ARROW_RELEASE_O3 to skip the downgrade: the bit-packing kernels are
# built around the inlining and unrolling -O3 enables, and the
# throughput figures quoted for them were measured with it in effect.
set(C_RELEASE_FLAGS "")
if(CMAKE_C_FLAGS_RELEASE MATCHES "-O3")
if(CMAKE_C_FLAGS_RELEASE MATCHES "-O3" AND NOT ARROW_RELEASE_O3)
string(APPEND C_RELEASE_FLAGS " -O2")
endif()
set(CXX_RELEASE_FLAGS "")
if(CMAKE_CXX_FLAGS_RELEASE MATCHES "-O3")
if(CMAKE_CXX_FLAGS_RELEASE MATCHES "-O3" AND NOT ARROW_RELEASE_O3)
string(APPEND CXX_RELEASE_FLAGS " -O2")
endif()
set(C_RELWITHDEBINFO_FLAGS "")
if(CMAKE_C_FLAGS_RELWITHDEBINFO MATCHES "-O3")
if(CMAKE_C_FLAGS_RELWITHDEBINFO MATCHES "-O3" AND NOT ARROW_RELEASE_O3)
string(APPEND C_RELWITHDEBINFO_FLAGS " -O2")
endif()
set(CXX_RELWITHDEBINFO_FLAGS "")
if(CMAKE_CXX_FLAGS_RELWITHDEBINFO MATCHES "-O3")
if(CMAKE_CXX_FLAGS_RELWITHDEBINFO MATCHES "-O3" AND NOT ARROW_RELEASE_O3)
string(APPEND CXX_RELWITHDEBINFO_FLAGS " -O2")
endif()
if(CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
Expand Down
25 changes: 25 additions & 0 deletions cpp/src/arrow/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -551,6 +551,7 @@ set(ARROW_UTIL_SRCS
util/decimal.cc
util/delimiting.cc
util/dict_util.cc
util/fastlanes/interleaved_pfor_baseline.cc
util/fixed_width_internal.cc
util/float16.cc
util/formatting.cc
Expand All @@ -566,6 +567,8 @@ set(ARROW_UTIL_SRCS
util/math_internal.cc
util/memory.cc
util/mutex.cc
util/pfor/pfor.cc
util/pfor/pfor_wrapper.cc
util/ree_util.cc
util/secure_string.cc
util/string.cc
Expand All @@ -591,6 +594,28 @@ append_runtime_avx512_src(ARROW_UTIL_SRCS util/bpacking_simd_avx512.cc)
append_runtime_sve128_src(ARROW_UTIL_SRCS util/bpacking_simd_128_alt.cc)
append_runtime_sve256_src(ARROW_UTIL_SRCS util/bpacking_simd_256.cc)

# One source compiled once per instruction set, the way bpacking_simd_256.cc is
# registered for both AVX2 and SVE256 above: the file forks on the platform
# macros, and only one of them is ever defined on a given target.
append_runtime_avx2_src(ARROW_UTIL_SRCS util/fastlanes/interleaved_pfor_simd.cc)
append_runtime_sve128_src(ARROW_UTIL_SRCS util/fastlanes/interleaved_pfor_simd.cc)

# The interleaved kernels are portable C++ with no intrinsics, so what they
# compile to is decided by the optimizer, not by the source. At width 16 gcc 11.5
# emits 289 instructions and no vector operations at -O2, 322 with 81 vector
# operations at Release's -O2 -ftree-vectorize, and 849 with 513 at -O3; the
# published throughput figures for this layout were all taken at the last of
# those. APPEND rather than a plain set, because the two calls above have already
# put the instruction-set flags in this property. Release only -- Debug and the
# sanitizer builds want their own level, and MSVC spells this differently.
if(NOT MSVC)
set_property(SOURCE util/fastlanes/interleaved_pfor_baseline.cc
util/fastlanes/interleaved_pfor_simd.cc
APPEND
PROPERTY COMPILE_OPTIONS
"$<$<OR:$<CONFIG:Release>,$<CONFIG:RelWithDebInfo>>:-O3>")
endif()

if(ARROW_WITH_BROTLI)
list(APPEND ARROW_UTIL_SRCS util/compression_brotli.cc)
endif()
Expand Down
2 changes: 2 additions & 0 deletions cpp/src/arrow/meson.build
Original file line number Diff line number Diff line change
Expand Up @@ -204,6 +204,8 @@ arrow_util_srcs = [
'util/math_internal.cc',
'util/memory.cc',
'util/mutex.cc',
'util/pfor/pfor.cc',
'util/pfor/pfor_wrapper.cc',
'util/ree_util.cc',
'util/secure_string.cc',
'util/string.cc',
Expand Down
4 changes: 4 additions & 0 deletions cpp/src/arrow/util/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -118,6 +118,10 @@ add_arrow_test(threading-utility-test
test_common.cc
thread_pool_test.cc)

add_arrow_test(pfor-test SOURCES pfor/pfor_test.cc)

add_arrow_benchmark(pfor/pfor_benchmark)

add_arrow_benchmark(bit_block_counter_benchmark)
add_arrow_benchmark(bit_util_benchmark)
add_arrow_benchmark(bitmap_reader_benchmark)
Expand Down
39 changes: 38 additions & 1 deletion cpp/src/arrow/util/bpacking.cc
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,29 @@ struct UnpackDynamicFunction {
ARROW_DISPATCH_TARGET_SVE256(&bpacking::unpack_sve256<Uint>) //
ARROW_DISPATCH_TARGET_SSE4_2(&bpacking::unpack_sse4_2<Uint>) //
ARROW_DISPATCH_TARGET_AVX2(&bpacking::unpack_avx2<Uint>) //
ARROW_DISPATCH_TARGET_AVX512(&bpacking::unpack_avx512<Uint>) //
// Cap bit-unpack dispatch at 256 bits. The generated AVX-512 kernels
// assemble vectors from scalar loads and use out-of-line calls, making
// them slower than the AVX2 path. Re-enable this target when its kernels
// use vector loads directly.
// ARROW_DISPATCH_TARGET_AVX512(&bpacking::unpack_avx512<Uint>) //
};
}
};

template <typename Uint>
struct UnpackBiasDynamicFunction {
using FunctionType = decltype(&bpacking::unpack_bias_scalar<Uint>);

static constexpr auto targets() {
return std::array{
ARROW_DISPATCH_TARGET_NONE(&bpacking::unpack_bias_scalar<Uint>) //
ARROW_DISPATCH_TARGET_NEON(&bpacking::unpack_bias_neon<Uint>) //
ARROW_DISPATCH_TARGET_SVE128(&bpacking::unpack_bias_sve128<Uint>) //
ARROW_DISPATCH_TARGET_SVE256(&bpacking::unpack_bias_sve256<Uint>) //
ARROW_DISPATCH_TARGET_SSE4_2(&bpacking::unpack_bias_sse4_2<Uint>) //
ARROW_DISPATCH_TARGET_AVX2(&bpacking::unpack_bias_avx2<Uint>) //
// Capped at 256 bits for the reason given in UnpackDynamicFunction above.
// ARROW_DISPATCH_TARGET_AVX512(&bpacking::unpack_bias_avx512<Uint>) //
};
}
};
Expand All @@ -57,4 +79,19 @@ template void unpack<uint16_t>(const uint8_t*, uint16_t*, const UnpackOptions&);
template void unpack<uint32_t>(const uint8_t*, uint32_t*, const UnpackOptions&);
template void unpack<uint64_t>(const uint8_t*, uint64_t*, const UnpackOptions&);

template <typename Uint>
void unpack_bias(const uint8_t* in, Uint* out, const UnpackOptions& opts, Uint bias) {
static const DynamicDispatch<UnpackBiasDynamicFunction<Uint>> dispatch;
return dispatch(in, out, opts, bias);
}

template void unpack_bias<uint8_t>(const uint8_t*, uint8_t*, const UnpackOptions&,
uint8_t);
template void unpack_bias<uint16_t>(const uint8_t*, uint16_t*, const UnpackOptions&,
uint16_t);
template void unpack_bias<uint32_t>(const uint8_t*, uint32_t*, const UnpackOptions&,
uint32_t);
template void unpack_bias<uint64_t>(const uint8_t*, uint64_t*, const UnpackOptions&,
uint64_t);

} // namespace arrow::internal
Loading
Loading