From 336fdb35868af7ce46001b73703dd3f7f2e39b8d Mon Sep 17 00:00:00 2001 From: Sreesh Maheshwar Date: Sun, 24 May 2026 23:25:58 +0100 Subject: [PATCH 001/512] GH-47435: [Python][Parquet] Add direct key encryption/decryption API (#49667) ### Rationale for this change See https://github.com/apache/arrow/issues/47435. ### What changes are included in this PR? Adds direct encryption / decryption Python API ### Are these changes tested? Yes, see PR. ### Are there any user-facing changes? Yes, new Python bindings. * GitHub Issue: #47435 Authored-by: Sreesh Maheshwar Signed-off-by: Adam Reeve --- docs/source/python/api/formats.rst | 2 + python/pyarrow/_parquet_encryption.pxd | 5 + python/pyarrow/_parquet_encryption.pyx | 209 ++++++++++++++++ python/pyarrow/includes/libparquet.pxd | 25 +- python/pyarrow/parquet/encryption.py | 4 +- .../pyarrow/tests/parquet/test_encryption.py | 226 ++++++++++++++++++ 6 files changed, 469 insertions(+), 2 deletions(-) diff --git a/docs/source/python/api/formats.rst b/docs/source/python/api/formats.rst index a4f02084c4a6..57a5e824fab1 100644 --- a/docs/source/python/api/formats.rst +++ b/docs/source/python/api/formats.rst @@ -119,6 +119,8 @@ Encrypted Parquet Files KmsConnectionConfig EncryptionConfiguration DecryptionConfiguration + create_encryption_properties + create_decryption_properties .. _api.orc: diff --git a/python/pyarrow/_parquet_encryption.pxd b/python/pyarrow/_parquet_encryption.pxd index 48939fe277fe..1a12a6d6785e 100644 --- a/python/pyarrow/_parquet_encryption.pxd +++ b/python/pyarrow/_parquet_encryption.pxd @@ -20,6 +20,11 @@ from pyarrow.includes.common cimport * from pyarrow.includes.libparquet_encryption cimport * +from pyarrow.includes.libparquet cimport ( + CSecureString, + CFileDecryptionPropertiesBuilder, + CFileEncryptionPropertiesBuilder, +) from pyarrow._parquet cimport (ParquetCipher, CFileEncryptionProperties, CFileDecryptionProperties, diff --git a/python/pyarrow/_parquet_encryption.pyx b/python/pyarrow/_parquet_encryption.pyx index db6a6b56ac4c..7fe7fa7491dc 100644 --- a/python/pyarrow/_parquet_encryption.pyx +++ b/python/pyarrow/_parquet_encryption.pyx @@ -711,3 +711,212 @@ cdef shared_ptr[CDecryptionConfiguration] pyarrow_unwrap_decryptionconfig(object if isinstance(decryptionconfig, DecryptionConfiguration): return ( decryptionconfig).unwrap() raise TypeError("Expected DecryptionConfiguration, got %s" % type(decryptionconfig)) + + +def create_decryption_properties( + footer_key, + *, + aad_prefix=None, + bint check_footer_integrity=True, + bint allow_plaintext_files=False, +): + """ + Create FileDecryptionProperties using a direct footer key. + + This is a low-level API that constructs decryption properties directly + from a plaintext key, bypassing the KMS-based :class:`CryptoFactory`. + It is intended for callers that manage key wrapping and storage + themselves (e.g. an application-level scheme). + + For most use cases, prefer the higher-level :class:`CryptoFactory` + with :class:`DecryptionConfiguration`, which implements the full + Parquet key management specification and is interoperable with + other tools and frameworks. + + .. note:: + Currently only uniform encryption (single key for footer and all + columns) is supported with this method. Per-column keys are not + yet available; files encrypted with per-column keys cannot be + decrypted using this function. + + Parameters + ---------- + footer_key : bytes + The decryption key for the file footer and all columns (uniform + encryption). Must be 16, 24, or 32 bytes for AES-128, AES-192, + or AES-256 respectively. + aad_prefix : bytes, optional + Additional Authenticated Data prefix. Must match the AAD prefix + that was used during encryption. Required if the AAD prefix was + not stored in the file metadata during encryption. + check_footer_integrity : bool, default True + Whether to verify footer integrity using the signature stored + in the file. Set to False only for debugging. + allow_plaintext_files : bool, default False + Whether to allow reading plaintext (unencrypted) files with + these decryption properties without raising an error. + + Returns + ------- + FileDecryptionProperties + Properties that can be passed to :func:`~pyarrow.parquet.read_table`, + :class:`~pyarrow.parquet.ParquetFile`, or + :class:`~pyarrow.dataset.ParquetFragmentScanOptions`. + + Examples + -------- + >>> import pyarrow.parquet as pq + >>> import pyarrow.parquet.encryption as pe + >>> props = pe.create_decryption_properties( + ... footer_key=b'0123456789abcdef', + ... aad_prefix=b'table_id', + ... ) + >>> table = pq.read_table('encrypted.parquet', decryption_properties=props) + """ + cdef: + c_string c_footer_key_str + CSecureString c_footer_key + CFileDecryptionPropertiesBuilder builder + shared_ptr[CFileDecryptionProperties] props + + if not isinstance(footer_key, bytes): + raise TypeError( + f"footer_key must be bytes, not {type(footer_key).__name__}" + ) + if len(footer_key) not in (16, 24, 32): + raise ValueError( + f"footer_key must be 16, 24, or 32 bytes, got {len(footer_key)}" + ) + + c_footer_key_str = footer_key + c_footer_key = CSecureString(move(c_footer_key_str)) + builder.footer_key(c_footer_key) + + if aad_prefix is not None: + if not isinstance(aad_prefix, bytes): + raise TypeError( + f"aad_prefix must be bytes, not {type(aad_prefix).__name__}" + ) + builder.aad_prefix(aad_prefix) + + if not check_footer_integrity: + builder.disable_footer_signature_verification() + + if allow_plaintext_files: + builder.plaintext_files_allowed() + + props = builder.build() + + return FileDecryptionProperties.wrap(props) + + +def create_encryption_properties( + footer_key, + *, + aad_prefix=None, + bint store_aad_prefix=True, + encryption_algorithm="AES_GCM_V1", + bint plaintext_footer=False, +): + """ + Create FileEncryptionProperties using a direct footer key. + + This is a low-level API that constructs encryption properties directly + from a plaintext key, bypassing the KMS-based :class:`CryptoFactory`. + It is intended for callers that manage key wrapping and storage + themselves (e.g. an application-level scheme). + + .. warning:: + The caller is responsible for key management best practices. + Reusing the same key for multiple files without unique data keys + weakens AES-GCM security. The higher-level :class:`CryptoFactory` + with :class:`EncryptionConfiguration` handles this automatically + and is interoperable with other tools and frameworks -- + prefer it unless you have a specific reason to manage + keys yourself. + + .. note:: + Currently only uniform encryption (single key for footer and all + columns) is supported with this method. Per-column keys are not + yet available; the provided key encrypts both the footer and + every column. + + Parameters + ---------- + footer_key : bytes + The encryption key for the file footer and all columns (uniform + encryption). Must be 16, 24, or 32 bytes for AES-128, AES-192, + or AES-256 respectively. + aad_prefix : bytes, optional + Additional Authenticated Data prefix for cryptographic binding. + store_aad_prefix : bool, default True + Whether to store the AAD prefix in the Parquet file metadata. + Set to False when the AAD prefix will be supplied externally + at read time. + Only meaningful when *aad_prefix* is provided. + encryption_algorithm : str, default "AES_GCM_V1" + Encryption algorithm. Either ``"AES_GCM_V1"`` or + ``"AES_GCM_CTR_V1"``. + plaintext_footer : bool, default False + Whether to leave the file footer unencrypted. When True, file + schema and metadata are readable without a key. + + Returns + ------- + FileEncryptionProperties + Properties that can be passed to :func:`~pyarrow.parquet.write_table` or + :class:`~pyarrow.parquet.ParquetWriter`. + + Examples + -------- + >>> import pyarrow as pa + >>> import pyarrow.parquet as pq + >>> import pyarrow.parquet.encryption as pe + >>> table = pa.table({'col': [1, 2, 3]}) + >>> props = pe.create_encryption_properties( + ... footer_key=b'0123456789abcdef', + ... aad_prefix=b'table_id', + ... store_aad_prefix=False, + ... ) + >>> pq.write_table(table, 'encrypted.parquet', encryption_properties=props) + """ + cdef: + c_string c_footer_key_str + CSecureString c_footer_key + CFileEncryptionPropertiesBuilder* builder + shared_ptr[CFileEncryptionProperties] props + ParquetCipher cipher + + if not isinstance(footer_key, bytes): + raise TypeError( + f"footer_key must be bytes, not {type(footer_key).__name__}" + ) + if len(footer_key) not in (16, 24, 32): + raise ValueError( + f"footer_key must be 16, 24, or 32 bytes, got {len(footer_key)}" + ) + + cipher = cipher_from_name(encryption_algorithm) + c_footer_key_str = footer_key + c_footer_key = CSecureString(move(c_footer_key_str)) + builder = new CFileEncryptionPropertiesBuilder(c_footer_key) + try: + builder.algorithm(cipher) + + if aad_prefix is not None: + if not isinstance(aad_prefix, bytes): + raise TypeError( + f"aad_prefix must be bytes, not {type(aad_prefix).__name__}" + ) + builder.aad_prefix(aad_prefix) + if not store_aad_prefix: + builder.disable_aad_prefix_storage() + + if plaintext_footer: + builder.set_plaintext_footer() + + props = builder.build() + finally: + del builder + + return FileEncryptionProperties.wrap(props) diff --git a/python/pyarrow/includes/libparquet.pxd b/python/pyarrow/includes/libparquet.pxd index a834bd5dfa0c..df353cc7805f 100644 --- a/python/pyarrow/includes/libparquet.pxd +++ b/python/pyarrow/includes/libparquet.pxd @@ -22,7 +22,8 @@ from pyarrow.includes.libarrow cimport (Type, CChunkedArray, CScalar, CSchema, CStatus, CTable, CMemoryPool, CBuffer, CKeyValueMetadata, CRandomAccessFile, COutputStream, CCacheOptions, - TimeUnit, CRecordBatchReader) + TimeUnit, CRecordBatchReader, + CSecureString) cdef extern from "parquet/api/schema.h" namespace "parquet::schema" nogil: @@ -635,6 +636,28 @@ cdef extern from "parquet/encryption/encryption.h" namespace "parquet" nogil: " parquet::FileDecryptionProperties": pass + cdef cppclass CFileDecryptionPropertiesBuilder\ + " parquet::FileDecryptionProperties::Builder": + CFileDecryptionPropertiesBuilder() except + + CFileDecryptionPropertiesBuilder* footer_key( + CSecureString footer_key) except + + CFileDecryptionPropertiesBuilder* aad_prefix( + c_string aad_prefix) except + + CFileDecryptionPropertiesBuilder* disable_footer_signature_verification() except + + CFileDecryptionPropertiesBuilder* plaintext_files_allowed() except + + shared_ptr[CFileDecryptionProperties] build() except + + cdef cppclass CFileEncryptionProperties\ " parquet::FileEncryptionProperties": pass + + cdef cppclass CFileEncryptionPropertiesBuilder\ + " parquet::FileEncryptionProperties::Builder": + CFileEncryptionPropertiesBuilder(CSecureString footer_key) except + + CFileEncryptionPropertiesBuilder* set_plaintext_footer() except + + CFileEncryptionPropertiesBuilder* algorithm( + ParquetCipher parquet_cipher) except + + CFileEncryptionPropertiesBuilder* aad_prefix( + c_string aad_prefix) except + + CFileEncryptionPropertiesBuilder* disable_aad_prefix_storage() except + + shared_ptr[CFileEncryptionProperties] build() except + diff --git a/python/pyarrow/parquet/encryption.py b/python/pyarrow/parquet/encryption.py index df6eed913fa5..ce95e5d45075 100644 --- a/python/pyarrow/parquet/encryption.py +++ b/python/pyarrow/parquet/encryption.py @@ -20,4 +20,6 @@ EncryptionConfiguration, DecryptionConfiguration, KmsConnectionConfig, - KmsClient) + KmsClient, + create_encryption_properties, + create_decryption_properties) diff --git a/python/pyarrow/tests/parquet/test_encryption.py b/python/pyarrow/tests/parquet/test_encryption.py index 4e2fb069bd06..6a3842f3edf8 100644 --- a/python/pyarrow/tests/parquet/test_encryption.py +++ b/python/pyarrow/tests/parquet/test_encryption.py @@ -37,6 +37,11 @@ COL_KEY = b"1234567890123450" COL_KEY_NAME = "col_key" +DIRECT_KEY_128 = b"0123456789abcdef" +DIRECT_KEY_192 = b"0123456789abcdef01234567" +DIRECT_KEY_256 = b"0123456789abcdef0123456789abcdef" +DIRECT_AAD_PREFIX = b"test_aad_prefix" + # Marks all of the tests in this module # Ignore these with pytest ... -m 'not parquet_encryption' @@ -722,3 +727,224 @@ def test_encrypted_parquet_read_table(tempdir, data_table, basic_encryption_conf result_table = pq.read_table( tempdir, decryption_properties=file_decryption_properties) assert data_table.equals(result_table) + + +class TestDirectKeyEncryption: + """Tests for create_encryption_properties / create_decryption_properties.""" + + @pytest.mark.parametrize("key", [ + DIRECT_KEY_128, DIRECT_KEY_192, DIRECT_KEY_256, + ], ids=["aes128", "aes192", "aes256"]) + def test_roundtrip_key_sizes(self, tempdir, data_table, key): + path = tempdir / f"direct_{len(key) * 8}.parquet" + + enc_props = pe.create_encryption_properties(footer_key=key) + pq.write_table(data_table, path, encryption_properties=enc_props) + + dec_props = pe.create_decryption_properties(footer_key=key) + result = pq.read_table(path, decryption_properties=dec_props) + assert data_table.equals(result) + + def test_roundtrip_with_aad_prefix(self, tempdir, data_table): + path = tempdir / "direct_aad.parquet" + + enc_props = pe.create_encryption_properties( + footer_key=DIRECT_KEY_128, + aad_prefix=DIRECT_AAD_PREFIX, + ) + pq.write_table(data_table, path, encryption_properties=enc_props) + + dec_props = pe.create_decryption_properties( + footer_key=DIRECT_KEY_128, + aad_prefix=DIRECT_AAD_PREFIX, + ) + result = pq.read_table(path, decryption_properties=dec_props) + assert data_table.equals(result) + + def test_roundtrip_aad_prefix_not_stored(self, tempdir, data_table): + """When store_aad_prefix=False, reader must supply aad_prefix.""" + path = tempdir / "direct_aad_not_stored.parquet" + + enc_props = pe.create_encryption_properties( + footer_key=DIRECT_KEY_128, + aad_prefix=DIRECT_AAD_PREFIX, + store_aad_prefix=False, + ) + pq.write_table(data_table, path, encryption_properties=enc_props) + + # Reading without aad_prefix should fail + dec_props_no_aad = pe.create_decryption_properties( + footer_key=DIRECT_KEY_128, + ) + with pytest.raises(IOError, match="AAD"): + pq.read_table(path, decryption_properties=dec_props_no_aad) + + # Reading with correct aad_prefix should succeed + dec_props = pe.create_decryption_properties( + footer_key=DIRECT_KEY_128, + aad_prefix=DIRECT_AAD_PREFIX, + ) + result = pq.read_table(path, decryption_properties=dec_props) + assert data_table.equals(result) + + def test_wrong_aad_prefix_fails(self, tempdir, data_table): + path = tempdir / "direct_wrong_aad.parquet" + + enc_props = pe.create_encryption_properties( + footer_key=DIRECT_KEY_128, + aad_prefix=DIRECT_AAD_PREFIX, + ) + pq.write_table(data_table, path, encryption_properties=enc_props) + + dec_props = pe.create_decryption_properties( + footer_key=DIRECT_KEY_128, + aad_prefix=b"wrong_prefix", + ) + with pytest.raises(IOError, match="AAD"): + pq.read_table(path, decryption_properties=dec_props) + + def test_encrypted_file_has_pare_magic(self, tempdir, data_table): + path = tempdir / "direct_magic.parquet" + + enc_props = pe.create_encryption_properties( + footer_key=DIRECT_KEY_128) + pq.write_table(data_table, path, encryption_properties=enc_props) + + with open(path, "rb") as f: + magic = f.read(4) + assert magic == b"PARE" + + def test_plaintext_footer(self, tempdir, data_table): + path = tempdir / "direct_plaintext_footer.parquet" + + enc_props = pe.create_encryption_properties( + footer_key=DIRECT_KEY_128, + plaintext_footer=True, + ) + pq.write_table(data_table, path, encryption_properties=enc_props) + + dec_props = pe.create_decryption_properties( + footer_key=DIRECT_KEY_128) + result = pq.read_table(path, decryption_properties=dec_props) + assert data_table.equals(result) + + def test_aes_gcm_ctr_v1_algorithm(self, tempdir, data_table): + path = tempdir / "direct_ctr.parquet" + + enc_props = pe.create_encryption_properties( + footer_key=DIRECT_KEY_128, + encryption_algorithm="AES_GCM_CTR_V1", + ) + pq.write_table(data_table, path, encryption_properties=enc_props) + + dec_props = pe.create_decryption_properties( + footer_key=DIRECT_KEY_128) + result = pq.read_table(path, decryption_properties=dec_props) + assert data_table.equals(result) + + def test_wrong_key_fails(self, tempdir, data_table): + path = tempdir / "direct_wrong_key.parquet" + + enc_props = pe.create_encryption_properties( + footer_key=DIRECT_KEY_128) + pq.write_table(data_table, path, encryption_properties=enc_props) + + wrong_key = b"fedcba9876543210" + dec_props = pe.create_decryption_properties(footer_key=wrong_key) + with pytest.raises(IOError, match="decrypt"): + pq.read_table(path, decryption_properties=dec_props) + + def test_reading_without_decryption_fails(self, tempdir, data_table): + path = tempdir / "direct_no_decrypt.parquet" + + enc_props = pe.create_encryption_properties( + footer_key=DIRECT_KEY_128) + pq.write_table(data_table, path, encryption_properties=enc_props) + + with pytest.raises(IOError, match="encrypted metadata"): + pq.read_table(path) + + def test_allow_plaintext_files(self, tempdir, data_table): + """Plaintext file reads should work when allow_plaintext_files=True.""" + path = tempdir / "plaintext.parquet" + pq.write_table(data_table, path) + + dec_props = pe.create_decryption_properties( + footer_key=DIRECT_KEY_128, + allow_plaintext_files=True, + ) + result = pq.read_table(path, decryption_properties=dec_props) + assert data_table.equals(result) + + def test_plaintext_file_rejected_by_default(self, tempdir, data_table): + """Default allow_plaintext_files=False should reject plaintext files.""" + path = tempdir / "plaintext_rejected.parquet" + pq.write_table(data_table, path) + + dec_props = pe.create_decryption_properties( + footer_key=DIRECT_KEY_128) + with pytest.raises(IOError, match="plaintext"): + pq.read_table(path, decryption_properties=dec_props) + + def test_check_footer_integrity_false(self, tempdir, data_table): + """check_footer_integrity=False should still allow decryption.""" + path = tempdir / "direct_no_footer_check.parquet" + + enc_props = pe.create_encryption_properties( + footer_key=DIRECT_KEY_128) + pq.write_table(data_table, path, encryption_properties=enc_props) + + dec_props = pe.create_decryption_properties( + footer_key=DIRECT_KEY_128, + check_footer_integrity=False, + ) + result = pq.read_table(path, decryption_properties=dec_props) + assert data_table.equals(result) + + def test_plaintext_footer_has_par1_magic(self, tempdir, data_table): + """plaintext_footer=True should produce PAR1 magic, not PARE.""" + path = tempdir / "direct_plaintext_magic.parquet" + + enc_props = pe.create_encryption_properties( + footer_key=DIRECT_KEY_128, + plaintext_footer=True, + ) + pq.write_table(data_table, path, encryption_properties=enc_props) + + with open(path, "rb") as f: + magic = f.read(4) + assert magic == b"PAR1" + + def test_invalid_key_length_raises(self): + with pytest.raises(ValueError, match="16, 24, or 32 bytes"): + pe.create_encryption_properties(footer_key=b"short") + + with pytest.raises(ValueError, match="16, 24, or 32 bytes"): + pe.create_encryption_properties(footer_key=b"") + + with pytest.raises(ValueError, match="16, 24, or 32 bytes"): + pe.create_decryption_properties(footer_key=b"short") + + def test_invalid_algorithm_raises(self): + with pytest.raises(ValueError, match="Invalid cipher name"): + pe.create_encryption_properties( + footer_key=DIRECT_KEY_128, + encryption_algorithm="INVALID", + ) + + def test_footer_key_rejects_non_bytes(self): + with pytest.raises(TypeError, match="footer_key must be bytes"): + pe.create_encryption_properties(footer_key="0123456789abcdef") + + with pytest.raises(TypeError, match="footer_key must be bytes"): + pe.create_decryption_properties(footer_key="0123456789abcdef") + + with pytest.raises(TypeError, match="footer_key must be bytes"): + pe.create_encryption_properties(footer_key=None) + + def test_aad_prefix_rejects_str(self, tempdir, data_table): + with pytest.raises(TypeError, match="aad_prefix must be bytes"): + pe.create_encryption_properties( + footer_key=DIRECT_KEY_128, + aad_prefix="not_bytes", + ) From 41b5fce9da2d0f8aabfbd785ba609fe15f857ad5 Mon Sep 17 00:00:00 2001 From: Sutou Kouhei Date: Mon, 25 May 2026 15:47:38 +0900 Subject: [PATCH 002/512] GH-50005: [C++] Use FetchContent for RapidJSON (#50006) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Rationale for this change If we use `ExternalProject` and `CMAKE_FIND_USE_PACKAGE_REGISTRY` https://cmake.org/cmake/help/latest/variable/CMAKE_FIND_USE_PACKAGE_REGISTRY.html and running `cmake` twice like https://github.com/apache/arrow/issues/48801#issuecomment-4498889561, `RapidJSONConfig.cmake` in `${BUILD_DIR}/rapidjson_ep/src/rapidjson_ep-install/lib/cmake/RapidjSON/`. It's not intentional. ### What changes are included in this PR? Use `FetchContent` instead of `ExternalProject`. ### Are these changes tested? Yes. ### Are there any user-facing changes? Yes. * GitHub Issue: #50005 Lead-authored-by: Sutou Kouhei Co-authored-by: Sutou Kouhei Co-authored-by: Raúl Cumplido Signed-off-by: Raúl Cumplido --- cpp/cmake_modules/ThirdpartyToolchain.cmake | 47 ++++++++++----------- 1 file changed, 23 insertions(+), 24 deletions(-) diff --git a/cpp/cmake_modules/ThirdpartyToolchain.cmake b/cpp/cmake_modules/ThirdpartyToolchain.cmake index d6256d9363ed..c08d0c4292b2 100644 --- a/cpp/cmake_modules/ThirdpartyToolchain.cmake +++ b/cpp/cmake_modules/ThirdpartyToolchain.cmake @@ -2782,34 +2782,33 @@ if(ARROW_BUILD_BENCHMARKS) FALSE) endif() -macro(build_rapidjson) - message(STATUS "Building RapidJSON from source") - set(RAPIDJSON_PREFIX - "${CMAKE_CURRENT_BINARY_DIR}/rapidjson_ep/src/rapidjson_ep-install") - set(RAPIDJSON_CMAKE_ARGS - ${EP_COMMON_CMAKE_ARGS} - -DRAPIDJSON_BUILD_DOC=OFF - -DRAPIDJSON_BUILD_EXAMPLES=OFF - -DRAPIDJSON_BUILD_TESTS=OFF - "-DCMAKE_INSTALL_PREFIX=${RAPIDJSON_PREFIX}") - - externalproject_add(rapidjson_ep - ${EP_COMMON_OPTIONS} - PREFIX "${CMAKE_BINARY_DIR}" - URL ${RAPIDJSON_SOURCE_URL} - URL_HASH "SHA256=${ARROW_RAPIDJSON_BUILD_SHA256_CHECKSUM}" - CMAKE_ARGS ${RAPIDJSON_CMAKE_ARGS}) +function(build_rapidjson) + list(APPEND CMAKE_MESSAGE_INDENT "RapidJSON: ") + message(STATUS "Building from source") - set(RAPIDJSON_INCLUDE_DIR "${RAPIDJSON_PREFIX}/include") - # The include directory must exist before it is referenced by a target. - file(MAKE_DIRECTORY "${RAPIDJSON_INCLUDE_DIR}") + fetchcontent_declare(rapidjson + ${FC_DECLARE_COMMON_OPTIONS} OVERRIDE_FIND_PACKAGE + URL ${RAPIDJSON_SOURCE_URL} + URL_HASH "SHA256=${ARROW_RAPIDJSON_BUILD_SHA256_CHECKSUM}") + prepare_fetchcontent() + set(LIB_INSTALL_DIR "${CMAKE_INSTALL_PREFIX}/lib") + set(RAPIDJSON_BUILD_DOC OFF) + set(RAPIDJSON_BUILD_EXAMPLES OFF) + set(RAPIDJSON_BUILD_TESTS OFF) + fetchcontent_makeavailable(rapidjson) + if(CMAKE_VERSION VERSION_LESS 3.28) + set_property(DIRECTORY ${rapidjson_SOURCE_DIR} PROPERTY EXCLUDE_FROM_ALL TRUE) + endif() add_library(RapidJSON INTERFACE IMPORTED) - target_include_directories(RapidJSON INTERFACE "${RAPIDJSON_INCLUDE_DIR}") - add_dependencies(RapidJSON rapidjson_ep) + target_include_directories(RapidJSON INTERFACE "${rapidjson_SOURCE_DIR}/include") + add_dependencies(RapidJSON rapidjson) - set(RAPIDJSON_VENDORED TRUE) -endmacro() + set(RAPIDJSON_VENDORED + TRUE + PARENT_SCOPE) + list(POP_BACK CMAKE_MESSAGE_INDENT) +endfunction() if(ARROW_WITH_RAPIDJSON) set(ARROW_RAPIDJSON_REQUIRED_VERSION "1.1.0") From 4ea1ad2afb27d8de78331a52e3cacab15b5c058b Mon Sep 17 00:00:00 2001 From: Sutou Kouhei Date: Mon, 25 May 2026 17:59:08 +0900 Subject: [PATCH 003/512] GH-50022: [Dev] Enable auto GitHub Copilot review (#50023) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Rationale for this change GitHub Copilot may reduce required review resource by committers. ### What changes are included in this PR? Add a configuration to `.asf.yaml`. See also: https://github.com/apache/infrastructure-asfyaml#copilot_code_review ### Are these changes tested? No. ### Are there any user-facing changes? No. * GitHub Issue: #50022 Authored-by: Sutou Kouhei Signed-off-by: Raúl Cumplido --- .asf.yaml | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/.asf.yaml b/.asf.yaml index 6c2a38388dda..2779249c7135 100644 --- a/.asf.yaml +++ b/.asf.yaml @@ -37,6 +37,11 @@ github: - hiroyuki-sato - HyukjinKwon + copilot_code_review: + enabled: true + review_drafts: true + review_on_push: true + notifications: commits: commits@arrow.apache.org issues_status: issues@arrow.apache.org From cd1811be25ce2c485ad00f3be0bf0ac44ea7b423 Mon Sep 17 00:00:00 2001 From: Kuinox Date: Mon, 25 May 2026 16:46:07 +0200 Subject: [PATCH 004/512] GH-48254: [Python][Parquet] Support extension types in read_schema (#48255) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Rationale for this change pq.read_schema drops extension types (UUID comes back as fixed_size_binary[16]), while ParquetFile.schema_arrow and read_table preserve them. Schema inspection via metadata should match table/extension behavior. ### What changes are included in this PR? - Plumb arrow_extensions_enabled into read_schema and return schema_arrow when enabled so extension types are preserved. - Add regression test ensuring UUID extension types are retained by read_schema and downgraded to binary(16) when extensions are disabled. ### Are these changes tested? - Yes: added unit test test_read_schema_uuid_extension_type ### Are there any user-facing changes? - Behavior improvement: read_schema now preserves extension types (e.g., UUID) when extensions are enabled; no API break Notes: - I don't know if the fact the column types being returned are now extension instead of fixed_size_binary[16], is considered a breaking change. - This PR patch was AI generated, but I personally reviewed it, the scope is small, and it looks fine to me. * GitHub Issue: #48254 Authored-by: Nicolas Vandeginste Signed-off-by: Raúl Cumplido --- python/pyarrow/parquet/core.py | 34 +++++++++++++------ python/pyarrow/tests/parquet/test_metadata.py | 22 ++++++++++++ 2 files changed, 46 insertions(+), 10 deletions(-) diff --git a/python/pyarrow/parquet/core.py b/python/pyarrow/parquet/core.py index 080bfa55c234..ff880fdcf52c 100644 --- a/python/pyarrow/parquet/core.py +++ b/python/pyarrow/parquet/core.py @@ -265,9 +265,9 @@ class ParquetFile: page_checksum_verification : bool, default False If True, verify the checksum for each page read from the file. arrow_extensions_enabled : bool, default True - If True, read Parquet logical types as Arrow extension types where possible, - (e.g., read JSON as the canonical `arrow.json` extension type or UUID as - the canonical `arrow.uuid` extension type). + If True, read Parquet logical types as Arrow extension types where + possible (e.g., read JSON as the canonical `arrow.json` extension type + or UUID as the canonical `arrow.uuid` extension type). Examples -------- @@ -2372,7 +2372,7 @@ def write_metadata(schema, where, metadata_collector=None, filesystem=None, def read_metadata(where, memory_map=False, decryption_properties=None, - filesystem=None): + filesystem=None, arrow_extensions_enabled=True): """ Read FileMetaData from footer of a single Parquet file. @@ -2387,6 +2387,10 @@ def read_metadata(where, memory_map=False, decryption_properties=None, If nothing passed, will be inferred based on path. Path will try to be found in the local on-disk filesystem otherwise it will be parsed as an URI to determine the filesystem. + arrow_extensions_enabled : bool, default True + If True, read Parquet logical types as Arrow extension types where + possible (e.g. UUID as the canonical `arrow.uuid` extension type). + If False, use the underlying storage types instead. Returns ------- @@ -2416,13 +2420,17 @@ def read_metadata(where, memory_map=False, decryption_properties=None, file_ctx = where = filesystem.open_input_file(where) with file_ctx: - file = ParquetFile(where, memory_map=memory_map, - decryption_properties=decryption_properties) + file = ParquetFile( + where, + memory_map=memory_map, + decryption_properties=decryption_properties, + arrow_extensions_enabled=arrow_extensions_enabled, + ) return file.metadata def read_schema(where, memory_map=False, decryption_properties=None, - filesystem=None): + filesystem=None, arrow_extensions_enabled=True): """ Read effective Arrow schema from Parquet file metadata. @@ -2437,6 +2445,9 @@ def read_schema(where, memory_map=False, decryption_properties=None, If nothing passed, will be inferred based on path. Path will try to be found in the local on-disk filesystem otherwise it will be parsed as an URI to determine the filesystem. + arrow_extensions_enabled : bool, default True + If True, read Parquet logical types as Arrow extension types where + possible (e.g., UUID as the canonical `arrow.uuid` extension type). Returns ------- @@ -2462,9 +2473,12 @@ def read_schema(where, memory_map=False, decryption_properties=None, with file_ctx: file = ParquetFile( - where, memory_map=memory_map, - decryption_properties=decryption_properties) - return file.schema.to_arrow_schema() + where, + memory_map=memory_map, + decryption_properties=decryption_properties, + arrow_extensions_enabled=arrow_extensions_enabled, + ) + return file.schema_arrow __all__ = ( diff --git a/python/pyarrow/tests/parquet/test_metadata.py b/python/pyarrow/tests/parquet/test_metadata.py index 857024da225c..665c9a0e8e49 100644 --- a/python/pyarrow/tests/parquet/test_metadata.py +++ b/python/pyarrow/tests/parquet/test_metadata.py @@ -849,3 +849,25 @@ def msg(c): with pytest.raises(TypeError, match=msg("FileMetaData")): pq.FileMetaData() + + +def test_read_schema_uuid_extension_type(tmp_path): + # These are the raw 16-byte payloads for + # UUID("e460f970-8351-474e-ac7f-a4673e4ba8cb").bytes and + # UUID("1e741495-eed5-43ea-9bd7-73dc91424baf").bytes. + data = [ + b'\xe4`\xf9p\x83QGN\xac\x7f\xa4g>K\xa8\xcb', + b'\x1et\x14\x95\xee\xd5C\xea\x9b\xd7s\xdc\x91BK\xaf', + None, + ] + table = pa.table([pa.array(data, type=pa.uuid())], names=["ext"]) + + file_path = tmp_path / "uuid.parquet" + file_path_str = str(file_path) + pq.write_table(table, file_path_str, store_schema=False) + + schema_default = pq.read_schema(file_path_str) + assert schema_default.field("ext").type == pa.uuid() + + schema_disabled = pq.read_schema(file_path_str, arrow_extensions_enabled=False) + assert schema_disabled.field("ext").type == pa.binary(16) From 52a51ea291f6c945d8ed1531092fd2dd388b1ff6 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 26 May 2026 09:37:27 +0900 Subject: [PATCH 005/512] MINOR: [CI] Bump docker/login-action from 4.1.0 to 4.2.0 (#50039) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bumps [docker/login-action](https://github.com/docker/login-action) from 4.1.0 to 4.2.0.
Release notes

Sourced from docker/login-action's releases.

v4.2.0

Full Changelog: https://github.com/docker/login-action/compare/v4.1.0...v4.2.0

Commits
  • 650006c Merge pull request #960 from docker/dependabot/npm_and_yarn/aws-sdk-dependenc...
  • 99df1a3 chore: update generated content
  • 3ab375f build(deps): bump the aws-sdk-dependencies group across 1 directory with 2 up...
  • 39d8580 Merge pull request #970 from docker/dependabot/npm_and_yarn/docker/actions-to...
  • 4eefcd3 chore: update generated content
  • 56d092c build(deps): bump @​docker/actions-toolkit from 0.86.0 to 0.90.0
  • e2e31ca Merge pull request #976 from docker/dependabot/npm_and_yarn/actions/core-3.0.1
  • 0bced94 chore: update generated content
  • 3e75a0f build(deps): bump @​actions/core from 3.0.0 to 3.0.1
  • 365bebd Merge pull request #984 from docker/dependabot/github_actions/aws-actions/con...
  • Additional commits viewable in compare view

[![Dependabot compatibility score](https://dependabot-badges.githubapp.com/badges/compatibility_score?dependency-name=docker/login-action&package-manager=github_actions&previous-version=4.1.0&new-version=4.2.0)](https://docs.github.com/en/github/managing-security-vulnerabilities/about-dependabot-security-updates#about-compatibility-scores) Dependabot will resolve any conflicts with this PR as long as you don't alter it yourself. You can also trigger a rebase manually by commenting `@ dependabot rebase`. [//]: # (dependabot-automerge-start) [//]: # (dependabot-automerge-end) ---
Dependabot commands and options
You can trigger Dependabot actions by commenting on this PR: - `@ dependabot rebase` will rebase this PR - `@ dependabot recreate` will recreate this PR, overwriting any edits that have been made to it - `@ dependabot show ignore conditions` will show all of the ignore conditions of the specified dependency - `@ dependabot ignore this major version` will close this PR and stop Dependabot creating any more for this major version (unless you reopen the PR or upgrade to it yourself) - `@ dependabot ignore this minor version` will close this PR and stop Dependabot creating any more for this minor version (unless you reopen the PR or upgrade to it yourself) - `@ dependabot ignore this dependency` will close this PR and stop Dependabot creating any more for this dependency (unless you reopen the PR or upgrade to it yourself)
Authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> Signed-off-by: Sutou Kouhei --- .github/workflows/package_linux.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/package_linux.yml b/.github/workflows/package_linux.yml index d02baba975a0..b34ac8e48c58 100644 --- a/.github/workflows/package_linux.yml +++ b/.github/workflows/package_linux.yml @@ -232,7 +232,7 @@ jobs: rake version:update popd - name: Login to GitHub Container registry - uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0 + uses: docker/login-action@650006c6eb7dba73a995cc03b0b2d7f5ca915bee # v4.2.0 with: registry: ghcr.io username: ${{ github.actor }} From 3589dc2a5bcf77e087cd40bd502490f90eba8017 Mon Sep 17 00:00:00 2001 From: Zehua Zou Date: Tue, 26 May 2026 10:31:31 +0800 Subject: [PATCH 006/512] GH-50010: [C++][Parquet] Fix undefined behavior in TypedColumnWriterImpl::UpdateLevelHistogram (#50011) ### Rationale for this change Fix an undefined behavior. ### What changes are included in this PR? It fixes the undefined behavior by delaying the span construction. ### Are these changes tested? Yes. ### Are there any user-facing changes? No. * GitHub Issue: #50010 Authored-by: Zehua Zou Signed-off-by: Gang Wu --- cpp/src/parquet/column_writer.cc | 13 ++++++------- 1 file changed, 6 insertions(+), 7 deletions(-) diff --git a/cpp/src/parquet/column_writer.cc b/cpp/src/parquet/column_writer.cc index b3ed46ee2d28..653f28f64bde 100644 --- a/cpp/src/parquet/column_writer.cc +++ b/cpp/src/parquet/column_writer.cc @@ -1807,20 +1807,19 @@ class TypedColumnWriterImpl : public ColumnWriterImpl, return; } - auto add_levels = [](std::vector& level_histogram, - std::span levels, int16_t max_level) { + auto add_levels = [](std::vector& level_histogram, const int16_t* levels, + int64_t num_levels, int16_t max_level) { if (max_level == 0) { return; } ARROW_DCHECK_EQ(static_cast(max_level) + 1, level_histogram.size()); - ::parquet::UpdateLevelHistogram(levels, level_histogram); + std::span level_span{levels, static_cast(num_levels)}; + ::parquet::UpdateLevelHistogram(level_span, level_histogram); }; - add_levels(page_size_statistics_->definition_level_histogram, - {def_levels, static_cast(num_levels)}, + add_levels(page_size_statistics_->definition_level_histogram, def_levels, num_levels, descr_->max_definition_level()); - add_levels(page_size_statistics_->repetition_level_histogram, - {rep_levels, static_cast(num_levels)}, + add_levels(page_size_statistics_->repetition_level_histogram, rep_levels, num_levels, descr_->max_repetition_level()); } From 50eb001b1306a2ed2f95bb6d36dcdb823d0486a9 Mon Sep 17 00:00:00 2001 From: Joris Van den Bossche Date: Tue, 26 May 2026 17:29:40 +0200 Subject: [PATCH 007/512] GH-50041: [CI][Python] Make test_string_to_tzinfo_pytz_fallback more robust for platforms supporting lower case tz names (#50042) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Rationale for this change Fix failing test on CI as reported in https://github.com/apache/arrow/issues/50041, for a test added in https://github.com/apache/arrow/pull/49694 ### What changes are included in this PR? The test relies on the lower case "europe/brussels" time zone name not being recognized by `zoneinfo` (and it generally is by `pytz`). But apparently on some platforms, `zoneinfo` does accept this. Updating the test to first test this, and thus skip the test dynamically. * GitHub Issue: #50041 Authored-by: Joris Van den Bossche Signed-off-by: Raúl Cumplido --- python/pyarrow/tests/test_types.py | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/python/pyarrow/tests/test_types.py b/python/pyarrow/tests/test_types.py index 4a33d79223d0..9b5c5efe1c30 100644 --- a/python/pyarrow/tests/test_types.py +++ b/python/pyarrow/tests/test_types.py @@ -511,11 +511,16 @@ def test_string_to_tzinfo_prefer_zoneinfo_false(): assert result == pytz.FixedOffset(90) -@pytest.mark.skipif( - sys.platform == 'darwin', reason="macOS supports those lower-case names" -) def test_string_to_tzinfo_pytz_fallback(): pytz = pytest.importorskip("pytz") + + try: + zoneinfo.ZoneInfo("europe/brussels") + except zoneinfo.ZoneInfoNotFoundError: + pass + else: + pytest.skip("zoneinfo supports lower-case names on this platform") + result = pa.lib.string_to_tzinfo("europe/brussels") expected = pytz.timezone("Europe/Brussels") assert result == expected From b5ece7e8b5b90ea307c08ee76b1bf702009c9c4a Mon Sep 17 00:00:00 2001 From: tadeja Date: Tue, 26 May 2026 23:18:13 +0200 Subject: [PATCH 008/512] GH-49767: [CI][C++] Disable mold on Ubuntu 24.04 to work around mold#1247 (#50033) ### Rationale for this change Fixes #49767 - intermittent SIGSEGV in the ODBC Linux CI job, the crash occurs during `_dl_relocate_object` before `main()` : https://github.com/apache/arrow/issues/49767#issuecomment-4520516388 apt ships mold 2.30.0 on Ubuntu 24.04 'Noble Numbat'. mold 2.30.0 has a non-determinism bug in section placement that causes this crash ([rui314/mold#1247](https://github.com/rui314/mold/issues/1247)). Fixed upstream in [mold 2.31.0](https://github.com/rui314/mold/releases/tag/v2.31.0). ### What changes are included in this PR? Disable mold on Ubuntu 24.04 in `ubuntu-24.04-cpp.dockerfile` by setting `ARROW_USE_MOLD=OFF`. Re-enable if apt mold reaches 2.31+ - note that Ubuntu 24.04 'Noble Numbat' might never update to 2.31+ https://launchpad.net/ubuntu/noble/+source/mold - its mold release is over 2 years old. A follow-up issue tracking apt mold updates and other CI jobs using mold (more added with #49898/#49899) will be opened separately. ### Are these changes tested? Yes, ODBC Linux job runs complete successfully on fork: [1st](https://github.com/tadeja/arrow/actions/runs/26389905866/job/77676883090), [2nd](https://github.com/tadeja/arrow/actions/runs/26389905866/job/77684025003) , [3rd](https://github.com/tadeja/arrow/actions/runs/26389905866/job/77685765548) , [4th](https://github.com/tadeja/arrow/actions/runs/26389905866/job/77693037433) ### Are there any user-facing changes? No. * GitHub Issue: #49767 Authored-by: Tadeja Kadunc Signed-off-by: Sutou Kouhei --- ci/docker/ubuntu-24.04-cpp.dockerfile | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/ci/docker/ubuntu-24.04-cpp.dockerfile b/ci/docker/ubuntu-24.04-cpp.dockerfile index 074301b472dc..3a98491ba8c2 100644 --- a/ci/docker/ubuntu-24.04-cpp.dockerfile +++ b/ci/docker/ubuntu-24.04-cpp.dockerfile @@ -176,6 +176,11 @@ RUN /arrow/ci/scripts/install_sccache.sh unknown-linux-musl /usr/local/bin # provided by the distribution: # - Abseil is old and we require a version that has CRC32C # - opentelemetry-cpp-dev is not packaged +# +# Ubuntu 24.04 apt mold is 2.30.0 which has a non-determinism bug in section placement: +# https://github.com/rui314/mold/issues/1247 fixed in mold 2.31.0. +# See https://github.com/apache/arrow/issues/49767 +# Re-enable ARROW_USE_MOLD if 2.31+ is released for apt Ubuntu 24.04. ENV absl_SOURCE=BUNDLED \ ARROW_ACERO=ON \ ARROW_AZURE=ON \ @@ -197,7 +202,7 @@ ENV absl_SOURCE=BUNDLED \ ARROW_SUBSTRAIT=ON \ ARROW_USE_ASAN=OFF \ ARROW_USE_CCACHE=ON \ - ARROW_USE_MOLD=ON \ + ARROW_USE_MOLD=OFF \ ARROW_USE_UBSAN=OFF \ ARROW_WITH_BROTLI=ON \ ARROW_WITH_BZ2=ON \ From d8757f0a0948bdb9c60ef05e4baf5f59fcfb5a58 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ra=C3=BAl=20Cumplido?= Date: Wed, 27 May 2026 12:39:31 +0200 Subject: [PATCH 009/512] MINOR: [Dev] Update collaborators list (#50050) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Rationale for this change The collaborator list defines triage role. It helps having triage people. https://github.com/apache/arrow/commits?author=tadeja ### What changes are included in this PR? Add to the list a collaborator that could benefit from triage role. ### Are these changes tested? No ### Are there any user-facing changes? No Authored-by: Raúl Cumplido Signed-off-by: AlenkaF --- .asf.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/.asf.yaml b/.asf.yaml index 2779249c7135..6f3d69055962 100644 --- a/.asf.yaml +++ b/.asf.yaml @@ -36,6 +36,7 @@ github: - EnricoMi - hiroyuki-sato - HyukjinKwon + - tadeja copilot_code_review: enabled: true From 4e04d46c4ebb54d5e4e06ada1b9a0e59c4287930 Mon Sep 17 00:00:00 2001 From: Nic Crane Date: Wed, 27 May 2026 14:46:19 +0100 Subject: [PATCH 010/512] GH-49275: [Doc] Update docs to specify disclosure of AI on mailing list messages (#49277) ### Rationale for this change Replying to AI on the mailing list is annoying ### What changes are included in this PR? Specify people should disclose ### Are these changes tested? Nah ### Are there any user-facing changes? Just docs * GitHub Issue: #49275 Authored-by: Nic Crane Signed-off-by: Nic Crane --- CONTRIBUTING.md | 5 +++++ README.md | 4 ++++ docs/source/developers/guide/communication.rst | 5 +++++ docs/source/developers/overview.rst | 1 + 4 files changed, 15 insertions(+) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index dc7d7a2244a6..f23c18f326b0 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -64,6 +64,11 @@ publicly on the [arrow-dev mailing-list](https://mail-archives.apache.org/mod_mb You can also ask on the mailing-list, see above. +## Are you using AI coding tools? + +Please review our [AI-generated code guidance](https://arrow.apache.org/docs/dev/developers/overview.html#ai-generated-code) +before submitting AI-assisted contributions. + ## Further information Please read our [development documentation](https://arrow.apache.org/docs/developers/index.html) diff --git a/README.md b/README.md index 3f778ce6b8ea..7ba3f56fce38 100644 --- a/README.md +++ b/README.md @@ -91,6 +91,9 @@ on git main. Please read our latest [project contribution guide][4]. +If you are using AI coding tools, please review our +[AI-generated code guidance][7]. + ## Getting involved Even if you do not plan to contribute to Apache Arrow itself or Arrow @@ -114,3 +117,4 @@ We use [AWS][6] for some of the required infrastructure for the project. [4]: https://arrow.apache.org/docs/dev/developers/index.html [5]: https://runs-on.com/ [6]: https://aws.amazon.com/ +[7]: https://arrow.apache.org/docs/dev/developers/overview.html#ai-generated-code diff --git a/docs/source/developers/guide/communication.rst b/docs/source/developers/guide/communication.rst index e87c64a2fe47..c22daeac79c0 100644 --- a/docs/source/developers/guide/communication.rst +++ b/docs/source/developers/guide/communication.rst @@ -119,6 +119,11 @@ It is announced on the development mailing list together with the link to join. .. seealso:: More information and links for subscribing to the mailing lists `can be found here `_. +.. note:: + Using AI tools to help with phrasing or grammar is absolutely fine. However, + we'd love to hear from *you* rather than your AI assistant - please don't + use AI to generate questions or comments on your behalf. + Recommended learning resources ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ diff --git a/docs/source/developers/overview.rst b/docs/source/developers/overview.rst index 0b3660e4d741..e91d1c28aca2 100644 --- a/docs/source/developers/overview.rst +++ b/docs/source/developers/overview.rst @@ -179,6 +179,7 @@ the following: * Watch for AI's tendency to generate overly verbose comments, unnecessary test cases, and incorrect fixes * Break down large PRs into smaller ones to make review easier +* AI agents should never tag or ping maintainers PR authors are also responsible for disclosing any copyrighted materials in submitted contributions. See the `ASF's guidance on AI-generated code From 2c1293f3c5438202e133a59d9478cecef33b35ea Mon Sep 17 00:00:00 2001 From: AnkitAhlawat Date: Thu, 28 May 2026 08:10:56 +0530 Subject: [PATCH 011/512] GH-47447: [C++] Fix replace_with_mask for null type arrays (#49950) ### Rationale for this change replace_with_mask crashes when called with null type arrays because a code path incorrectly returns Status::OK() from a function with return type Result. This PR fixes the issue by ensuring the function returns a valid int64_t value instead of Status::OK() for successful execution. ### What changes are included in this PR? 1. Fixed the replace_with_mask kernel implementation for null type arrays. 2. Replaced the invalid Status::OK() return path with a valid int64_t result. 3. Prevented construction of Result with an OK Status. ### Are these changes tested? Yes, manually run the test case ### Are there any user-facing changes? No **This PR contains a "Critical Fix".** This change fixes a crash in replace_with_mask when operating on null type arrays. The crash occurs due to an invalid successful return path internally constructing Result with Status::OK(). * GitHub Issue: #47447 Authored-by: Ankit.Ahlawat@ibm.com Signed-off-by: Sutou Kouhei --- .../arrow/compute/kernels/vector_replace.cc | 8 ++--- .../compute/kernels/vector_replace_test.cc | 36 +++++++++++++++++++ python/pyarrow/tests/test_compute.py | 28 +++++++++++++++ 3 files changed, 68 insertions(+), 4 deletions(-) diff --git a/cpp/src/arrow/compute/kernels/vector_replace.cc b/cpp/src/arrow/compute/kernels/vector_replace.cc index 1a3e743784d4..ae39977aaea5 100644 --- a/cpp/src/arrow/compute/kernels/vector_replace.cc +++ b/cpp/src/arrow/compute/kernels/vector_replace.cc @@ -217,15 +217,15 @@ struct ReplaceMaskImpl> { static Result ExecScalarMask(KernelContext* ctx, const ArraySpan& array, const BooleanScalar& mask, ExecValue replacements, int64_t replacements_offset, ExecResult* out) { - out->value = array; - return Status::OK(); + out->value = array.ToArrayData(); + return replacements_offset; } static Result ExecArrayMask(KernelContext* ctx, const ArraySpan& array, const ArraySpan& mask, int64_t mask_offset, ExecValue replacements, int64_t replacements_offset, ExecResult* out) { - out->value = array; - return Status::OK(); + out->value = array.ToArrayData(); + return replacements_offset; } }; diff --git a/cpp/src/arrow/compute/kernels/vector_replace_test.cc b/cpp/src/arrow/compute/kernels/vector_replace_test.cc index 587b9f2a60eb..9dc8e70ab6a2 100644 --- a/cpp/src/arrow/compute/kernels/vector_replace_test.cc +++ b/cpp/src/arrow/compute/kernels/vector_replace_test.cc @@ -199,6 +199,13 @@ class TestReplaceBoolean : public TestReplaceKernel { } }; +class TestReplaceNull : public TestReplaceKernel { + protected: + std::shared_ptr type() override { + return TypeTraits::type_singleton(); + } +}; + class TestReplaceFixedSizeBinary : public TestReplaceKernel { protected: std::shared_ptr type() override { return fixed_size_binary(3); } @@ -538,6 +545,35 @@ TEST_F(TestReplaceBoolean, ReplaceWithMask) { } } +TEST_F(TestReplaceNull, ReplaceWithMask) { + std::vector cases = { + {this->array("[]"), this->mask_scalar(false), this->array("[]"), this->array("[]")}, + {this->array("[]"), this->mask_scalar(true), this->array("[]"), this->array("[]")}, + {this->array("[]"), this->null_mask_scalar(), this->array("[]"), this->array("[]")}, + + {this->array("[null]"), this->mask_scalar(false), this->array("[]"), + this->array("[null]")}, + + {this->array("[null]"), this->mask_scalar(true), this->array("[null]"), + this->array("[null]")}, + + {this->array("[null]"), this->null_mask_scalar(), this->array("[]"), + this->array("[null]")}, + + {this->array("[null, null]"), this->mask("[false, false]"), this->array("[]"), + this->array("[null, null]")}, + {this->array("[null, null]"), this->mask("[true, true]"), + this->array("[null, null]"), this->array("[null, null]")}, + {this->array("[null, null]"), this->mask("[null, null]"), this->array("[]"), + this->array("[null, null]")}, + }; + + for (auto test_case : cases) { + this->Assert(ReplaceWithMask, test_case.input, test_case.mask, test_case.replacements, + test_case.expected); + } +} + TEST_F(TestReplaceBoolean, ReplaceWithMaskErrors) { EXPECT_RAISES_WITH_MESSAGE_THAT( Invalid, diff --git a/python/pyarrow/tests/test_compute.py b/python/pyarrow/tests/test_compute.py index 4e44a912d965..1c82d6c944bd 100644 --- a/python/pyarrow/tests/test_compute.py +++ b/python/pyarrow/tests/test_compute.py @@ -1269,6 +1269,34 @@ def test_extract_regex_span(): assert struct.tolist() == expected +def test_replace_with_mask_null_type(): + # GH-47447: replace_with_mask crashed for null type arrays + input = pa.array([None], pa.null()) + replacements = pa.array([None], pa.null()) + + result = pc.replace_with_mask(input, True, replacements) + assert result.type == pa.null() + result.validate(full=True) + assert result.to_pylist() == [None] + + result = pc.replace_with_mask(input, False, replacements) + assert result.type == pa.null() + result.validate(full=True) + assert result.to_pylist() == [None] + + mask = pa.array([True]) + result = pc.replace_with_mask(input, mask, replacements) + assert result.type == pa.null() + result.validate(full=True) + assert result.to_pylist() == [None] + + mask = pa.array([False]) + result = pc.replace_with_mask(input, mask, replacements) + assert result.type == pa.null() + result.validate(full=True) + assert result.to_pylist() == [None] + + def test_binary_join(): ar_list = pa.array([['foo', 'bar'], None, []]) expected = pa.array(['foo-bar', None, '']) From eb375a5c38b46ce96725f2f7f6376eba0e516e4f Mon Sep 17 00:00:00 2001 From: tadeja Date: Thu, 28 May 2026 04:53:41 +0200 Subject: [PATCH 012/512] GH-49776: [CI][C++] Install libc6-dbg in apt-based Linux C++ images (#50034) ### Rationale for this change Fixes #49776. `libc6-dbg` lets gdb resolve symbols inside `/lib*/ld-linux-x86-64.so.2` ### What changes are included in this PR? Add `libc6-dbg` to the apt-install list in `ci/docker/ubuntu-24.04-cpp.dockerfile`. Additionally add it also to `ci/docker/ubuntu-24.04-cpp-minimal.dockerfile`, `ci/docker/ubuntu-22.04-cpp.dockerfile`, `ci/docker/ubuntu-22.04-cpp-minimal.dockerfile`, `ci/docker/debian-13-cpp.dockerfile` and `ci/docker/debian-experimental-cpp.dockerfile`. ### Are these changes tested? Yes, already used to symbolize the [#49767](https://github.com/apache/arrow/issues/49767#issuecomment-4520516388). ### Are there any user-facing changes? No. * GitHub Issue: #49776 Authored-by: Tadeja Kadunc Signed-off-by: Sutou Kouhei --- ci/docker/debian-13-cpp.dockerfile | 1 + ci/docker/debian-experimental-cpp.dockerfile | 1 + ci/docker/ubuntu-22.04-cpp-minimal.dockerfile | 1 + ci/docker/ubuntu-22.04-cpp.dockerfile | 1 + ci/docker/ubuntu-24.04-cpp-minimal.dockerfile | 1 + ci/docker/ubuntu-24.04-cpp.dockerfile | 1 + 6 files changed, 6 insertions(+) diff --git a/ci/docker/debian-13-cpp.dockerfile b/ci/docker/debian-13-cpp.dockerfile index 92997fae4e0e..542c0446c59b 100644 --- a/ci/docker/debian-13-cpp.dockerfile +++ b/ci/docker/debian-13-cpp.dockerfile @@ -56,6 +56,7 @@ RUN apt-get update -y -q && \ libboost-system-dev \ libbrotli-dev \ libbz2-dev \ + libc6-dbg \ libcurl4-openssl-dev \ libgflags-dev \ libgmock-dev \ diff --git a/ci/docker/debian-experimental-cpp.dockerfile b/ci/docker/debian-experimental-cpp.dockerfile index 58b49eb70c96..3e3241d33a56 100644 --- a/ci/docker/debian-experimental-cpp.dockerfile +++ b/ci/docker/debian-experimental-cpp.dockerfile @@ -49,6 +49,7 @@ RUN if [ -n "${gcc}" ]; then \ libbrotli-dev \ libbz2-dev \ libc-ares-dev \ + libc6-dbg \ libcurl4-openssl-dev \ libgflags-dev \ libgmock-dev \ diff --git a/ci/docker/ubuntu-22.04-cpp-minimal.dockerfile b/ci/docker/ubuntu-22.04-cpp-minimal.dockerfile index d38dd418e296..80e97c440a45 100644 --- a/ci/docker/ubuntu-22.04-cpp-minimal.dockerfile +++ b/ci/docker/ubuntu-22.04-cpp-minimal.dockerfile @@ -31,6 +31,7 @@ RUN apt-get update -y -q && \ curl \ gdb \ git \ + libc6-dbg \ libssl-dev \ libcurl4-openssl-dev \ patch \ diff --git a/ci/docker/ubuntu-22.04-cpp.dockerfile b/ci/docker/ubuntu-22.04-cpp.dockerfile index 183bed4c6c8f..7b0182012e73 100644 --- a/ci/docker/ubuntu-22.04-cpp.dockerfile +++ b/ci/docker/ubuntu-22.04-cpp.dockerfile @@ -77,6 +77,7 @@ RUN apt-get update -y -q && \ libbrotli-dev \ libbz2-dev \ libc-ares-dev \ + libc6-dbg \ libcurl4-openssl-dev \ libgflags-dev \ libgmock-dev \ diff --git a/ci/docker/ubuntu-24.04-cpp-minimal.dockerfile b/ci/docker/ubuntu-24.04-cpp-minimal.dockerfile index 5e114d5dcd9f..904cfa99dfef 100644 --- a/ci/docker/ubuntu-24.04-cpp-minimal.dockerfile +++ b/ci/docker/ubuntu-24.04-cpp-minimal.dockerfile @@ -31,6 +31,7 @@ RUN apt-get update -y -q && \ curl \ gdb \ git \ + libc6-dbg \ libssl-dev \ libcurl4-openssl-dev \ patch \ diff --git a/ci/docker/ubuntu-24.04-cpp.dockerfile b/ci/docker/ubuntu-24.04-cpp.dockerfile index 3a98491ba8c2..c5fc190917fc 100644 --- a/ci/docker/ubuntu-24.04-cpp.dockerfile +++ b/ci/docker/ubuntu-24.04-cpp.dockerfile @@ -78,6 +78,7 @@ RUN apt-get update -y -q && \ libbrotli-dev \ libbz2-dev \ libc-ares-dev \ + libc6-dbg \ libcurl4-openssl-dev \ libgflags-dev \ libgmock-dev \ From d9bc3b947cb94fee783c2261d4090d1ed3dcf050 Mon Sep 17 00:00:00 2001 From: Nic Crane Date: Thu, 28 May 2026 10:04:56 +0100 Subject: [PATCH 013/512] GH-49901: [R] Bump minimum supported R version to 4.2 now that 4.6 is out (#49929) ### Rationale for this change We don't need to support R 4.1 as a new version is out ### What changes are included in this PR? Bump minimum supported version to 4.2 ### Are these changes tested? In CI, sure ### Are there any user-facing changes? Nope * GitHub Issue: #49901 Authored-by: Nic Crane Signed-off-by: Nic Crane --- dev/tasks/macros.jinja | 2 -- .../r/github.linux.arrow.version.back.compat.yml | 14 ++------------ dev/tasks/r/github.linux.versions.yml | 1 - r/DESCRIPTION | 2 +- 4 files changed, 3 insertions(+), 16 deletions(-) diff --git a/dev/tasks/macros.jinja b/dev/tasks/macros.jinja index 4edcfad3aecd..8e9a41b46b75 100644 --- a/dev/tasks/macros.jinja +++ b/dev/tasks/macros.jinja @@ -277,8 +277,6 @@ env: {# use filter to cast to string and convert to lowercase to match yaml boolean #} {% set is_fork = (not is_upstream_b)|lower %} -{% set r_release = {"ver": "4.2", "rt" : "42"} %} -{% set r_oldrel = {"ver": "4.1", "rt" : "40"} %} {%- macro github_set_env(env) -%} {% if env is defined %} diff --git a/dev/tasks/r/github.linux.arrow.version.back.compat.yml b/dev/tasks/r/github.linux.arrow.version.back.compat.yml index 7c2695f95638..bf5300710ebc 100644 --- a/dev/tasks/r/github.linux.arrow.version.back.compat.yml +++ b/dev/tasks/r/github.linux.arrow.version.back.compat.yml @@ -90,10 +90,8 @@ jobs: - { old_arrow_version: '10.0.1', r: '4.2' } - { old_arrow_version: '9.0.0', r: '4.2' } - { old_arrow_version: '8.0.0', r: '4.2' } - - { old_arrow_version: '7.0.0', r: '4.1' } - - { old_arrow_version: '6.0.1', r: '4.1' } - - { old_arrow_version: '5.0.0', r: '4.1' } - - { old_arrow_version: '4.0.0', r: '4.1' } + - { old_arrow_version: '7.0.0', r: '4.2' } + - { old_arrow_version: '6.0.1', r: '4.2' } env: ARROW_R_DEV: "TRUE" OLD_ARROW_VERSION: {{ '${{ matrix.config.old_arrow_version }}' }} @@ -102,17 +100,9 @@ jobs: - name: Prepare RSPM run: | - old_arrow_version_major=$(echo ${OLD_ARROW_VERSION} | cut -d. -f1) - if [ ${old_arrow_version_major} -ge 6 ]; then - # Binary arrow packages for Ubuntu 22.04 are available only - # for 6.0.0 or later. - rspm="https://packagemanager.rstudio.com/cran/__linux__/jammy/latest" - echo "RSPM=${rspm}" >> $GITHUB_ENV - else # testthat requires fs which now requires libuv1-dev. Install it # when RSPM isn't available sudo apt update && sudo apt install -y -V libuv1-dev - fi - uses: r-lib/actions/setup-r@v2 with: r-version: {{ '${{ matrix.config.r }}' }} diff --git a/dev/tasks/r/github.linux.versions.yml b/dev/tasks/r/github.linux.versions.yml index 612d84b185fe..644494bfbae6 100644 --- a/dev/tasks/r/github.linux.versions.yml +++ b/dev/tasks/r/github.linux.versions.yml @@ -30,7 +30,6 @@ jobs: r_version: # We test devel, release, and oldrel in regular CI. # This is for older versions - - "4.1" - "4.2" - "4.3" - "4.4" diff --git a/r/DESCRIPTION b/r/DESCRIPTION index b65c95accab0..d8be13cae859 100644 --- a/r/DESCRIPTION +++ b/r/DESCRIPTION @@ -22,7 +22,7 @@ Description: 'Apache' 'Arrow' is a cross-language language-independent columnar memory format for flat and hierarchical data, organized for efficient analytic operations on modern hardware. This package provides an interface to the 'Arrow C++' library. -Depends: R (>= 4.1) +Depends: R (>= 4.2) License: Apache License (>= 2.0) URL: https://github.com/apache/arrow/, https://arrow.apache.org/docs/r/ BugReports: https://github.com/apache/arrow/issues From ca8a194bd15df3c450e67415d40d2d222576849e Mon Sep 17 00:00:00 2001 From: metsw24-max Date: Fri, 29 May 2026 16:54:58 +0530 Subject: [PATCH 014/512] GH-50063: [C++] Validate buffer size for row-major tensors (#50064) ### Rationale for this change `ValidateTensorParameters` in cpp/src/arrow/tensor.cc only runs the `CheckTensorStridesValidity` buffer-overrun guard when strides are passed explicitly. With implicit (row-major) strides it computes strides for overflow but never checks the data buffer is large enough for the shape, so a tensor whose shape exceeds its buffer is accepted and later read out of bounds. This is reachable from IPC `ReadTensor`, where the shape comes from the flatbuffer and the body size is independent of it. ### What changes are included in this PR? Run `CheckTensorStridesValidity` on the computed row-major strides too. ### Are these changes tested? Added a case to `TestTensor.MakeFailureCases`. ### Are there any user-facing changes? No. **This PR contains a "Critical Fix".** Crafted IPC tensor metadata (or any caller building a row-major tensor over an undersized buffer) bypassed the bounds check, enabling an out-of-bounds read. * GitHub Issue: #50063 Authored-by: metsw24-max Signed-off-by: Rok Mihevc --- cpp/src/arrow/tensor.cc | 1 + cpp/src/arrow/tensor_test.cc | 3 +++ 2 files changed, 4 insertions(+) diff --git a/cpp/src/arrow/tensor.cc b/cpp/src/arrow/tensor.cc index 8cdf7f82d264..49b7a0b28ecf 100644 --- a/cpp/src/arrow/tensor.cc +++ b/cpp/src/arrow/tensor.cc @@ -216,6 +216,7 @@ Status ValidateTensorParameters(const std::shared_ptr& type, std::vector tmp_strides; RETURN_NOT_OK(ComputeRowMajorStrides(checked_cast(*type), shape, &tmp_strides)); + RETURN_NOT_OK(CheckTensorStridesValidity(data, shape, tmp_strides, type)); } if (dim_names.size() > shape.size()) { return Status::Invalid("too many dim_names are supplied"); diff --git a/cpp/src/arrow/tensor_test.cc b/cpp/src/arrow/tensor_test.cc index 05dcf38e1e61..2a2f564e7910 100644 --- a/cpp/src/arrow/tensor_test.cc +++ b/cpp/src/arrow/tensor_test.cc @@ -268,6 +268,9 @@ TEST(TestTensor, MakeFailureCases) { ASSERT_RAISES(Invalid, Tensor::Make(float64(), data, shape, {sizeof(double) * 12, sizeof(double)})); + // row-major (implicit strides) shape larger than the backing buffer + ASSERT_RAISES(Invalid, Tensor::Make(float64(), data, {3, 100})); + // too many dim_names are supplied ASSERT_RAISES(Invalid, Tensor::Make(float64(), data, shape, {}, {"foo", "bar", "baz"})); } From 35eed28c6faba1f5b24f6c0dac0973345ee44171 Mon Sep 17 00:00:00 2001 From: Antoine Pitrou Date: Mon, 1 Jun 2026 10:42:45 +0200 Subject: [PATCH 015/512] GH-49946: [Format] Better document equivalence between IPC file and streams (#49947) ### Rationale for this change As discussed in https://lists.apache.org/thread/jpxl3yzm96wkxzb1clokxklsy32b3plh, we want to better document the rough equivalence between IPC files and streams. ### What changes are included in this PR? * Recommendation around emitting "nice" IPC file footers that don't reorder, omit or repeat batches * Better outlining of deviations of the IPC file format vs. the IPC streaming format * Assorted minor wording and presentation improvements in the columnar / IPC doc ### Are these changes tested? N/A. ### Are there any user-facing changes? No. * GitHub Issue: #49946 Lead-authored-by: Antoine Pitrou Co-authored-by: Antoine Pitrou Co-authored-by: Andrew Lamb Signed-off-by: Antoine Pitrou --- docs/source/format/Columnar.rst | 159 ++++++++++++++++++++------------ 1 file changed, 100 insertions(+), 59 deletions(-) diff --git a/docs/source/format/Columnar.rst b/docs/source/format/Columnar.rst index cf7ff0c3dec6..2e81bd0f9424 100644 --- a/docs/source/format/Columnar.rst +++ b/docs/source/format/Columnar.rst @@ -1209,8 +1209,8 @@ of these types: * DictionaryBatch We specify a so-called *encapsulated IPC message* format which -includes a serialized Flatbuffer type along with an optional message -body. We define this message format before describing how to serialize +includes a serialized Flatbuffers metadata message along with an optional +message body. We define this message format before describing how to serialize each constituent IPC message type. .. _ipc-message-format: @@ -1238,7 +1238,7 @@ The encapsulated binary message format is as follows: Schematically, we have: :: - + @@ -1248,8 +1248,8 @@ can be relocated between streams. Otherwise the amount of padding between the metadata and the message body could be non-deterministic. The ``metadata_size`` includes the size of the ``Message`` plus -padding. The ``metadata_flatbuffer`` contains a serialized ``Message`` -Flatbuffer value, which internally includes: +padding. The ``metadata_flatbuffer`` contains a Flatbuffers-serialized +``Message`` value, which internally includes: * A version number * A particular message value (one of ``Schema``, ``RecordBatch``, or @@ -1264,11 +1264,11 @@ can be read. Schema message -------------- -The Flatbuffers files `Schema.fbs`_ contains the definitions for all +The Flatbuffers definition file `Schema.fbs`_ contains the definitions for all built-in data types and the ``Schema`` metadata type which represents the schema of a given record batch. A schema consists of an ordered sequence of fields, each having a name and type. A serialized ``Schema`` -does not contain any data buffers, only type metadata. +does not contain any body, only metadata. The ``Field`` Flatbuffers type contains the metadata for a single array. This includes: @@ -1287,7 +1287,7 @@ array. This includes: We additionally provide both schema-level and field-level ``custom_metadata`` attributes allowing for systems to insert their -own application defined metadata to customize behavior. +own application-defined metadata to customize behavior. .. _ipc-recordbatch-message: @@ -1302,17 +1302,18 @@ thus no memory copying. The serialized form of the record batch is the following: -* The ``data header``, defined as the ``RecordBatch`` type in - `Message.fbs`_. -* The ``body``, a flat sequence of memory buffers written end-to-end - with appropriate padding to ensure a minimum of 8-byte alignment +* The metadata describing the record batch layout, defined as the ``RecordBatch`` + type in `Message.fbs`_. +* The record batch body, a flat sequence of binary buffers written end-to-end, + with appropriate padding between each of them to ensure a minimum of 8-byte + alignment. -The data header contains the following: +The record batch metadata contains the following: * The length and null count for each flattened field in the record - batch + batch. * The memory offset and length of each constituent ``Buffer`` in the - record batch's body + record batch's body. Fields and buffers are flattened by a pre-order depth-first traversal of the fields in the record batch. For example, let's consider the @@ -1333,22 +1334,21 @@ The flattened version of this is: :: For the buffers produced, we would have the following (refer to the table above): :: - buffer 0: field 0 validity - buffer 1: field 1 validity - buffer 2: field 1 values - buffer 3: field 2 validity - buffer 4: field 2 offsets - buffer 5: field 3 validity - buffer 6: field 3 values - buffer 7: field 4 validity - buffer 8: field 4 values - buffer 9: field 5 validity - buffer 10: field 5 offsets - buffer 11: field 5 data - -The ``Buffer`` Flatbuffers value describes the location and size of a -piece of memory. Generally these are interpreted relative to the -**encapsulated message format** defined below. + buffer 0: field 0 ('col1') validity + buffer 1: field 1 ('col1.a') validity + buffer 2: field 1 ('col1.a') values + buffer 3: field 2 ('col1.b') validity + buffer 4: field 2 ('col1.b') offsets + buffer 5: field 3 ('col1.b.item') validity + buffer 6: field 3 ('col1.b.item') values + buffer 7: field 4 ('col1.c') validity + buffer 8: field 4 ('col1.c') values + buffer 9: field 5 ('col2') validity + buffer 10: field 5 ('col2') offsets + buffer 11: field 5 ('col2') data + +The ``Buffer`` Flatbuffers value describes the location and size of a buffer's +data, relative to the start of the RecordBatch message's body. The ``size`` field of ``Buffer`` is not required to account for padding bytes. Since this metadata can be used to communicate in-memory pointer @@ -1391,7 +1391,6 @@ have two entries in each RecordBatch. For a RecordBatch of this schema with buffer 12: col2 data buffer 13: col2 data - Compression ----------- @@ -1452,18 +1451,19 @@ serialized form is as follows: ratios. Byte Order (`Endianness`_) ---------------------------- +-------------------------- + +The Arrow IPC format is little-endian by default. -The Arrow format is little endian by default. +Serialized Schema metadata has an endianness field indicating the endianness of +Arrow data in all the RecordBatch and DictionaryBatch message bodies. +Typically this is the endianness of the system where the Arrow data was generated. +While some IPC reader implementations may allow for byte-swapping when reading +an IPC Stream or File with non-native endianness, other implementations may simply +refuse reading such data. -Serialized Schema metadata has an endianness field indicating -endianness of RecordBatches. Typically this is the endianness of the -system where the RecordBatch was generated. The main use case is -exchanging RecordBatches between systems with the same Endianness. At -first we will return an error when trying to read a Schema with an -endianness that does not match the underlying system. The reference -implementation is focused on Little Endian and provides tests for -it. Eventually we may provide automatic conversion via byte swapping. +Note that the endianness field only applies to RecordBatch and DictionaryBatch +message bodies, not to message metadata or any other signaling in the IPC formats. IPC Streaming Format -------------------- @@ -1483,7 +1483,7 @@ a ``RecordBatch`` it should be defined in a ``DictionaryBatch``. :: ... - + ... ... @@ -1502,9 +1502,14 @@ message flatbuffer is read, you can then read the message body. The stream writer can signal end-of-stream (EOS) either by writing 8 bytes containing the 4-byte continuation indicator (``0xFFFFFFFF``) followed by 0 -metadata length (``0x00000000``) or closing the stream interface. We -recommend the ".arrows" file extension for the streaming format although -in many cases these streams will not ever be stored as files. +metadata length (``0x00000000``) or closing the stream interface. + +File extension and MIME type +~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +IPC Streams are not typically stored as files, but when they are, we recommend +the ".arrows" file extension. The registered MIME type for IPC Streams is +`vnd.apache.arrow.stream`_. IPC File Format --------------- @@ -1524,21 +1529,55 @@ Schematically we have: ::