From 05af96236192efb9dbfee40447add09f0ef878c2 Mon Sep 17 00:00:00 2001 From: 1fanwang <1fannnw@gmail.com> Date: Wed, 23 Sep 2026 23:39:22 -0700 Subject: [PATCH 1/4] GH-51043: Reject null required PyArrow objects Signed-off-by: 1fanwang <1fannnw@gmail.com> --- python/pyarrow/_dataset.pyx | 5 ++++- python/pyarrow/_parquet.pyx | 5 +++-- python/pyarrow/array.pxi | 5 +++-- python/pyarrow/tests/parquet/test_metadata.py | 6 ++++++ python/pyarrow/tests/test_array.py | 4 ++++ python/pyarrow/tests/test_dataset.py | 6 ++++++ 6 files changed, 26 insertions(+), 5 deletions(-) diff --git a/python/pyarrow/_dataset.pyx b/python/pyarrow/_dataset.pyx index d40614a61fc0..e06fc6caf6c9 100644 --- a/python/pyarrow/_dataset.pyx +++ b/python/pyarrow/_dataset.pyx @@ -1121,7 +1121,8 @@ cdef class FileSystemDataset(Dataset): cdef: CFileSystemDataset* filesystem_dataset - def __init__(self, fragments, Schema schema, FileFormat format, + def __init__(self, fragments, Schema schema not None, + FileFormat format not None, FileSystem filesystem=None, root_partition=None): cdef: FileFragment fragment=None @@ -1138,6 +1139,8 @@ cdef class FileSystemDataset(Dataset): ) for fragment in fragments: + if fragment is None: + raise TypeError("Fragment must not be None") c_fragments.push_back( static_pointer_cast[CFileFragment, CFragment]( fragment.unwrap())) diff --git a/python/pyarrow/_parquet.pyx b/python/pyarrow/_parquet.pyx index 2621c3bd6d61..a1e2b1ba3003 100644 --- a/python/pyarrow/_parquet.pyx +++ b/python/pyarrow/_parquet.pyx @@ -755,7 +755,8 @@ cdef class SortingColumn: self.nulls_first = nulls_first @classmethod - def from_ordering(cls, Schema schema, sort_keys, null_placement='at_end'): + def from_ordering(cls, Schema schema not None, sort_keys, + null_placement='at_end'): """ Create a tuple of SortingColumn objects from the same arguments as :class:`pyarrow.compute.SortOptions`. @@ -816,7 +817,7 @@ cdef class SortingColumn: return tuple(sorting_columns) @staticmethod - def to_ordering(Schema schema, sorting_columns): + def to_ordering(Schema schema not None, sorting_columns): """ Convert a tuple of SortingColumn objects to the same format as :class:`pyarrow.compute.SortOptions`. diff --git a/python/pyarrow/array.pxi b/python/pyarrow/array.pxi index 865d2079fc0e..9de37b807909 100644 --- a/python/pyarrow/array.pxi +++ b/python/pyarrow/array.pxi @@ -4351,8 +4351,9 @@ cdef class DictionaryArray(Array): return self._indices @staticmethod - def from_buffers(DataType type, int64_t length, buffers, Array dictionary, - int64_t null_count=-1, int64_t offset=0): + def from_buffers(DataType type not None, int64_t length, buffers, + Array dictionary, int64_t null_count=-1, + int64_t offset=0): """ Construct a DictionaryArray from buffers. diff --git a/python/pyarrow/tests/parquet/test_metadata.py b/python/pyarrow/tests/parquet/test_metadata.py index 9eee70b125e5..eafffd3436e0 100644 --- a/python/pyarrow/tests/parquet/test_metadata.py +++ b/python/pyarrow/tests/parquet/test_metadata.py @@ -351,6 +351,12 @@ def test_parquet_sorting_column(): with pytest.raises(ValueError): pq.SortingColumn.from_ordering(schema, (("a", "not a valid sort order"))) + with pytest.raises(TypeError, match="Argument 'schema' has incorrect type"): + pq.SortingColumn.from_ordering(None, ()) + + with pytest.raises(TypeError, match="Argument 'schema' has incorrect type"): + pq.SortingColumn.to_ordering(None, ()) + with pytest.raises(ValueError, match="inconsistent null placement"): sorting_cols = ( pq.SortingColumn(1, nulls_first=True), diff --git a/python/pyarrow/tests/test_array.py b/python/pyarrow/tests/test_array.py index eb81a4409e8d..4033fd722bab 100644 --- a/python/pyarrow/tests/test_array.py +++ b/python/pyarrow/tests/test_array.py @@ -985,6 +985,10 @@ def test_dictionary_from_buffers(offset): offset=offset) assert a[offset:] == b + with pytest.raises(TypeError, match="Argument 'type' has incorrect type"): + pa.DictionaryArray.from_buffers( + None, len(a), a.indices.buffers(), a.dictionary) + @pytest.mark.numpy def test_dictionary_from_numpy(): diff --git a/python/pyarrow/tests/test_dataset.py b/python/pyarrow/tests/test_dataset.py index 0a94c0bd9875..49d6513b80fe 100644 --- a/python/pyarrow/tests/test_dataset.py +++ b/python/pyarrow/tests/test_dataset.py @@ -396,6 +396,12 @@ def test_filesystem_dataset(mockfs): # validation of required arguments with pytest.raises(TypeError, match="incorrect type"): ds.FileSystemDataset(fragments, file_format, schema) + with pytest.raises(TypeError, match="Fragment must not be None"): + ds.FileSystemDataset([None], schema=schema, format=file_format) + with pytest.raises(TypeError, match="Argument 'schema' has incorrect type"): + ds.FileSystemDataset(fragments, schema=None, format=file_format) + with pytest.raises(TypeError, match="Argument 'format' has incorrect type"): + ds.FileSystemDataset(fragments, schema=schema, format=None) # validation of root_partition with pytest.raises(TypeError, match="incorrect type"): ds.FileSystemDataset(fragments, schema=schema, From f03b3c8435e067f644ebe1c8679c7a4596d3e6a5 Mon Sep 17 00:00:00 2001 From: 1fanwang <1fannnw@gmail.com> Date: Wed, 23 Sep 2026 23:39:22 -0700 Subject: [PATCH 2/4] GH-51043: Reject null dictionary arrays Signed-off-by: 1fanwang <1fannnw@gmail.com> --- python/pyarrow/array.pxi | 2 +- python/pyarrow/tests/test_array.py | 3 +++ 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/python/pyarrow/array.pxi b/python/pyarrow/array.pxi index 9de37b807909..580f59a33204 100644 --- a/python/pyarrow/array.pxi +++ b/python/pyarrow/array.pxi @@ -4352,7 +4352,7 @@ cdef class DictionaryArray(Array): @staticmethod def from_buffers(DataType type not None, int64_t length, buffers, - Array dictionary, int64_t null_count=-1, + Array dictionary not None, int64_t null_count=-1, int64_t offset=0): """ Construct a DictionaryArray from buffers. diff --git a/python/pyarrow/tests/test_array.py b/python/pyarrow/tests/test_array.py index 4033fd722bab..dcd7e03a2ad0 100644 --- a/python/pyarrow/tests/test_array.py +++ b/python/pyarrow/tests/test_array.py @@ -988,6 +988,9 @@ def test_dictionary_from_buffers(offset): with pytest.raises(TypeError, match="Argument 'type' has incorrect type"): pa.DictionaryArray.from_buffers( None, len(a), a.indices.buffers(), a.dictionary) + with pytest.raises(TypeError, match="Argument 'dictionary' has incorrect type"): + pa.DictionaryArray.from_buffers( + a.type, len(a), a.indices.buffers(), None) @pytest.mark.numpy From 82ce629325bfa1c270ef93b19db06879ceea673d Mon Sep 17 00:00:00 2001 From: 1fanwang <1fannnw@gmail.com> Date: Wed, 23 Sep 2026 23:39:22 -0700 Subject: [PATCH 3/4] GH-51043: [Python] Validate required array buffer arguments Signed-off-by: 1fanwang <1fannnw@gmail.com> --- python/pyarrow/array.pxi | 10 ++++++---- python/pyarrow/tests/test_array.py | 24 ++++++++++++++++++++++++ 2 files changed, 30 insertions(+), 4 deletions(-) diff --git a/python/pyarrow/array.pxi b/python/pyarrow/array.pxi index 580f59a33204..66793c79b9b3 100644 --- a/python/pyarrow/array.pxi +++ b/python/pyarrow/array.pxi @@ -1334,8 +1334,8 @@ cdef class Array(_PandasConvertible): (_reduce_array_data(self.sp_array.get().data().get()),) @staticmethod - def from_buffers(DataType type, length, buffers, null_count=-1, offset=0, - children=None): + def from_buffers(DataType type not None, length, buffers, null_count=-1, + offset=0, children=None): """ Construct an Array from a sequence of buffers. @@ -1391,6 +1391,8 @@ cdef class Array(_PandasConvertible): c_buffers.push_back(pyarrow_unwrap_buffer(buf)) for child in children: + if child is None: + raise TypeError("Array child must not be None") c_child_data.push_back(child.ap.data()) array_data = CArrayData.MakeWithChildren(type.sp_type, length, @@ -4769,8 +4771,8 @@ cdef class RunEndEncodedArray(Array): run_ends, values, 0) @staticmethod - def from_buffers(DataType type, length, buffers, null_count=-1, offset=0, - children=None): + def from_buffers(DataType type not None, length, buffers, null_count=-1, + offset=0, children=None): """ Construct a RunEndEncodedArray from all the parameters that make up an Array. diff --git a/python/pyarrow/tests/test_array.py b/python/pyarrow/tests/test_array.py index dcd7e03a2ad0..a0a7a41a7a6d 100644 --- a/python/pyarrow/tests/test_array.py +++ b/python/pyarrow/tests/test_array.py @@ -782,6 +782,9 @@ def test_array_from_buffers(): with pytest.raises(TypeError): pa.Array.from_buffers(pa.int16(), 3, ['', ''], offset=1) + with pytest.raises(TypeError, match="Argument 'type' has incorrect type"): + pa.Array.from_buffers(type=None, length=0, buffers=[]) + def test_string_binary_from_buffers(): array = pa.array(["a", None, "b", "c"]) @@ -818,6 +821,20 @@ def test_string_binary_from_buffers(): assert copied.null_count == 0 +@pytest.mark.parametrize("array_type", [pa.StringArray, pa.LargeStringArray]) +def test_string_from_buffers_null_buffers(array_type): + empty = array_type.from_buffers( + length=0, value_offsets=None, data=pa.py_buffer(b"")) + assert empty.to_pylist() == [] + + with pytest.raises(pa.ArrowInvalid, match="Value data buffer is null"): + array_type.from_buffers(length=0, value_offsets=None, data=None) + + with pytest.raises(pa.ArrowInvalid, match="Non-empty array but offsets are null"): + array_type.from_buffers( + length=1, value_offsets=None, data=pa.py_buffer(b"x")) + + def test_string_view_from_buffers(): array = pa.array( [ @@ -866,6 +883,10 @@ def test_list_from_buffers(list_type_factory): pa.Array.from_buffers(ty, 4, buffers[:ty.num_buffers], children=[child, child]) + with pytest.raises(TypeError, match="Array child must not be None"): + pa.Array.from_buffers( + type=ty, length=4, buffers=buffers[:ty.num_buffers], children=[None]) + def test_struct_from_buffers(): ty = pa.struct([pa.field('a', pa.int16()), pa.field('b', pa.utf8())]) @@ -4203,6 +4224,9 @@ def test_run_end_encoded_from_buffers(): pa.RunEndEncodedArray.from_buffers(ree_type, length, buffers, 1, offset, children) + with pytest.raises(TypeError, match="Argument 'type' has incorrect type"): + pa.RunEndEncodedArray.from_buffers(type=None, length=0, buffers=[]) + @pytest.mark.numpy def test_run_end_encoded_from_array_with_type(): From cafded2f07526c210dd3dda9b51de8218e186736 Mon Sep 17 00:00:00 2001 From: 1fanwang <1fannnw@gmail.com> Date: Wed, 23 Sep 2026 23:39:22 -0700 Subject: [PATCH 4/4] GH-51043: [Python] Reject None in more required object constructors Raise TypeError instead of aborting the interpreter when ScanNodeOptions, ProjectNodeOptions, ParquetSchema, ColumnSchema, or SubTreeFileSystem receive None for a required argument. Signed-off-by: 1fanwang <1fannnw@gmail.com> --- python/pyarrow/_acero.pyx | 2 ++ python/pyarrow/_dataset.pyx | 2 +- python/pyarrow/_fs.pyx | 2 +- python/pyarrow/_parquet.pyx | 4 ++-- python/pyarrow/tests/test_misc.py | 29 +++++++++++++++++++++++++++++ 5 files changed, 35 insertions(+), 4 deletions(-) diff --git a/python/pyarrow/_acero.pyx b/python/pyarrow/_acero.pyx index e27509469ef5..0a5285056931 100644 --- a/python/pyarrow/_acero.pyx +++ b/python/pyarrow/_acero.pyx @@ -141,6 +141,8 @@ cdef class _ProjectNodeOptions(ExecNodeOptions): vector[c_string] c_names for expr in expressions: + if expr is None: + raise TypeError("Expression must not be None") c_expressions.push_back(expr.unwrap()) if names is not None: diff --git a/python/pyarrow/_dataset.pyx b/python/pyarrow/_dataset.pyx index e06fc6caf6c9..f2f8788afd7b 100644 --- a/python/pyarrow/_dataset.pyx +++ b/python/pyarrow/_dataset.pyx @@ -4239,5 +4239,5 @@ class ScanNodeOptions(_ScanNodeOptions): Preserve implicit ordering of data. """ - def __init__(self, Dataset dataset, **kwargs): + def __init__(self, Dataset dataset not None, **kwargs): self._set_options(dataset, kwargs) diff --git a/python/pyarrow/_fs.pyx b/python/pyarrow/_fs.pyx index 0739b6acba32..10b6ac52dd55 100644 --- a/python/pyarrow/_fs.pyx +++ b/python/pyarrow/_fs.pyx @@ -1205,7 +1205,7 @@ cdef class SubTreeFileSystem(FileSystem): For usage of the methods see examples for :func:`~pyarrow.fs.LocalFileSystem`. """ - def __init__(self, base_path, FileSystem base_fs): + def __init__(self, base_path, FileSystem base_fs not None): cdef: c_string pathstr shared_ptr[CSubTreeFileSystem] wrapped diff --git a/python/pyarrow/_parquet.pyx b/python/pyarrow/_parquet.pyx index a1e2b1ba3003..ddb266d86712 100644 --- a/python/pyarrow/_parquet.pyx +++ b/python/pyarrow/_parquet.pyx @@ -1248,7 +1248,7 @@ cdef class FileMetaData(_Weakrefable): cdef class ParquetSchema(_Weakrefable): """A Parquet schema.""" - def __cinit__(self, FileMetaData container): + def __cinit__(self, FileMetaData container not None): self.parent = container self.schema = container._metadata.schema() @@ -1337,7 +1337,7 @@ cdef class ColumnSchema(_Weakrefable): ParquetSchema parent const ColumnDescriptor* descr - def __cinit__(self, ParquetSchema schema, int index): + def __cinit__(self, ParquetSchema schema not None, int index): self.parent = schema self.index = index # for pickling support self.descr = schema.schema.Column(index) diff --git a/python/pyarrow/tests/test_misc.py b/python/pyarrow/tests/test_misc.py index 856873f441b4..e7cc0a459e08 100644 --- a/python/pyarrow/tests/test_misc.py +++ b/python/pyarrow/tests/test_misc.py @@ -270,3 +270,32 @@ def test_extension_type_constructor_errors(klass): msg = f"Do not call {klass.__name__}'s constructor directly, use .* instead." with pytest.raises(TypeError, match=msg): klass() + + +@pytest.mark.processes +@pytest.mark.parametrize(("module", "call", "message"), [ + pytest.param( + "acero", "ScanNodeOptions(None)", "Argument 'dataset'", + marks=[pytest.mark.acero, pytest.mark.dataset], id="scan"), + pytest.param( + "acero", "ProjectNodeOptions([None])", "Expression must not be None", + marks=pytest.mark.acero, id="project"), + pytest.param( + "parquet", "ParquetSchema(None)", "Argument 'container'", + marks=pytest.mark.parquet, id="parquet-schema"), + pytest.param( + "parquet", "ColumnSchema(None, 0)", "Argument 'schema'", + marks=pytest.mark.parquet, id="column-schema"), + pytest.param( + "fs", 'SubTreeFileSystem("", None)', "Argument 'base_fs'", + id="subtree"), +]) +def test_required_arrow_objects_reject_none_without_crashing( + module: str, call: str, message: str +) -> None: + result = subprocess.run( + args=[sys.executable, "-c", f"import pyarrow.{module} as mod; mod.{call}"], + capture_output=True, text=True, timeout=30, + ) + assert result.returncode == 1, result.stderr + assert f"TypeError: {message}" in result.stderr