diff --git a/python/pyarrow/_acero.pyx b/python/pyarrow/_acero.pyx index e27509469ef5..0a5285056931 100644 --- a/python/pyarrow/_acero.pyx +++ b/python/pyarrow/_acero.pyx @@ -141,6 +141,8 @@ cdef class _ProjectNodeOptions(ExecNodeOptions): vector[c_string] c_names for expr in expressions: + if expr is None: + raise TypeError("Expression must not be None") c_expressions.push_back(expr.unwrap()) if names is not None: diff --git a/python/pyarrow/_dataset.pyx b/python/pyarrow/_dataset.pyx index d40614a61fc0..f2f8788afd7b 100644 --- a/python/pyarrow/_dataset.pyx +++ b/python/pyarrow/_dataset.pyx @@ -1121,7 +1121,8 @@ cdef class FileSystemDataset(Dataset): cdef: CFileSystemDataset* filesystem_dataset - def __init__(self, fragments, Schema schema, FileFormat format, + def __init__(self, fragments, Schema schema not None, + FileFormat format not None, FileSystem filesystem=None, root_partition=None): cdef: FileFragment fragment=None @@ -1138,6 +1139,8 @@ cdef class FileSystemDataset(Dataset): ) for fragment in fragments: + if fragment is None: + raise TypeError("Fragment must not be None") c_fragments.push_back( static_pointer_cast[CFileFragment, CFragment]( fragment.unwrap())) @@ -4236,5 +4239,5 @@ class ScanNodeOptions(_ScanNodeOptions): Preserve implicit ordering of data. """ - def __init__(self, Dataset dataset, **kwargs): + def __init__(self, Dataset dataset not None, **kwargs): self._set_options(dataset, kwargs) diff --git a/python/pyarrow/_fs.pyx b/python/pyarrow/_fs.pyx index 0739b6acba32..10b6ac52dd55 100644 --- a/python/pyarrow/_fs.pyx +++ b/python/pyarrow/_fs.pyx @@ -1205,7 +1205,7 @@ cdef class SubTreeFileSystem(FileSystem): For usage of the methods see examples for :func:`~pyarrow.fs.LocalFileSystem`. """ - def __init__(self, base_path, FileSystem base_fs): + def __init__(self, base_path, FileSystem base_fs not None): cdef: c_string pathstr shared_ptr[CSubTreeFileSystem] wrapped diff --git a/python/pyarrow/_parquet.pyx b/python/pyarrow/_parquet.pyx index 2621c3bd6d61..ddb266d86712 100644 --- a/python/pyarrow/_parquet.pyx +++ b/python/pyarrow/_parquet.pyx @@ -755,7 +755,8 @@ cdef class SortingColumn: self.nulls_first = nulls_first @classmethod - def from_ordering(cls, Schema schema, sort_keys, null_placement='at_end'): + def from_ordering(cls, Schema schema not None, sort_keys, + null_placement='at_end'): """ Create a tuple of SortingColumn objects from the same arguments as :class:`pyarrow.compute.SortOptions`. @@ -816,7 +817,7 @@ cdef class SortingColumn: return tuple(sorting_columns) @staticmethod - def to_ordering(Schema schema, sorting_columns): + def to_ordering(Schema schema not None, sorting_columns): """ Convert a tuple of SortingColumn objects to the same format as :class:`pyarrow.compute.SortOptions`. @@ -1247,7 +1248,7 @@ cdef class FileMetaData(_Weakrefable): cdef class ParquetSchema(_Weakrefable): """A Parquet schema.""" - def __cinit__(self, FileMetaData container): + def __cinit__(self, FileMetaData container not None): self.parent = container self.schema = container._metadata.schema() @@ -1336,7 +1337,7 @@ cdef class ColumnSchema(_Weakrefable): ParquetSchema parent const ColumnDescriptor* descr - def __cinit__(self, ParquetSchema schema, int index): + def __cinit__(self, ParquetSchema schema not None, int index): self.parent = schema self.index = index # for pickling support self.descr = schema.schema.Column(index) diff --git a/python/pyarrow/array.pxi b/python/pyarrow/array.pxi index 865d2079fc0e..66793c79b9b3 100644 --- a/python/pyarrow/array.pxi +++ b/python/pyarrow/array.pxi @@ -1334,8 +1334,8 @@ cdef class Array(_PandasConvertible): (_reduce_array_data(self.sp_array.get().data().get()),) @staticmethod - def from_buffers(DataType type, length, buffers, null_count=-1, offset=0, - children=None): + def from_buffers(DataType type not None, length, buffers, null_count=-1, + offset=0, children=None): """ Construct an Array from a sequence of buffers. @@ -1391,6 +1391,8 @@ cdef class Array(_PandasConvertible): c_buffers.push_back(pyarrow_unwrap_buffer(buf)) for child in children: + if child is None: + raise TypeError("Array child must not be None") c_child_data.push_back(child.ap.data()) array_data = CArrayData.MakeWithChildren(type.sp_type, length, @@ -4351,8 +4353,9 @@ cdef class DictionaryArray(Array): return self._indices @staticmethod - def from_buffers(DataType type, int64_t length, buffers, Array dictionary, - int64_t null_count=-1, int64_t offset=0): + def from_buffers(DataType type not None, int64_t length, buffers, + Array dictionary not None, int64_t null_count=-1, + int64_t offset=0): """ Construct a DictionaryArray from buffers. @@ -4768,8 +4771,8 @@ cdef class RunEndEncodedArray(Array): run_ends, values, 0) @staticmethod - def from_buffers(DataType type, length, buffers, null_count=-1, offset=0, - children=None): + def from_buffers(DataType type not None, length, buffers, null_count=-1, + offset=0, children=None): """ Construct a RunEndEncodedArray from all the parameters that make up an Array. diff --git a/python/pyarrow/tests/parquet/test_metadata.py b/python/pyarrow/tests/parquet/test_metadata.py index 9eee70b125e5..eafffd3436e0 100644 --- a/python/pyarrow/tests/parquet/test_metadata.py +++ b/python/pyarrow/tests/parquet/test_metadata.py @@ -351,6 +351,12 @@ def test_parquet_sorting_column(): with pytest.raises(ValueError): pq.SortingColumn.from_ordering(schema, (("a", "not a valid sort order"))) + with pytest.raises(TypeError, match="Argument 'schema' has incorrect type"): + pq.SortingColumn.from_ordering(None, ()) + + with pytest.raises(TypeError, match="Argument 'schema' has incorrect type"): + pq.SortingColumn.to_ordering(None, ()) + with pytest.raises(ValueError, match="inconsistent null placement"): sorting_cols = ( pq.SortingColumn(1, nulls_first=True), diff --git a/python/pyarrow/tests/test_array.py b/python/pyarrow/tests/test_array.py index eb81a4409e8d..a0a7a41a7a6d 100644 --- a/python/pyarrow/tests/test_array.py +++ b/python/pyarrow/tests/test_array.py @@ -782,6 +782,9 @@ def test_array_from_buffers(): with pytest.raises(TypeError): pa.Array.from_buffers(pa.int16(), 3, ['', ''], offset=1) + with pytest.raises(TypeError, match="Argument 'type' has incorrect type"): + pa.Array.from_buffers(type=None, length=0, buffers=[]) + def test_string_binary_from_buffers(): array = pa.array(["a", None, "b", "c"]) @@ -818,6 +821,20 @@ def test_string_binary_from_buffers(): assert copied.null_count == 0 +@pytest.mark.parametrize("array_type", [pa.StringArray, pa.LargeStringArray]) +def test_string_from_buffers_null_buffers(array_type): + empty = array_type.from_buffers( + length=0, value_offsets=None, data=pa.py_buffer(b"")) + assert empty.to_pylist() == [] + + with pytest.raises(pa.ArrowInvalid, match="Value data buffer is null"): + array_type.from_buffers(length=0, value_offsets=None, data=None) + + with pytest.raises(pa.ArrowInvalid, match="Non-empty array but offsets are null"): + array_type.from_buffers( + length=1, value_offsets=None, data=pa.py_buffer(b"x")) + + def test_string_view_from_buffers(): array = pa.array( [ @@ -866,6 +883,10 @@ def test_list_from_buffers(list_type_factory): pa.Array.from_buffers(ty, 4, buffers[:ty.num_buffers], children=[child, child]) + with pytest.raises(TypeError, match="Array child must not be None"): + pa.Array.from_buffers( + type=ty, length=4, buffers=buffers[:ty.num_buffers], children=[None]) + def test_struct_from_buffers(): ty = pa.struct([pa.field('a', pa.int16()), pa.field('b', pa.utf8())]) @@ -985,6 +1006,13 @@ def test_dictionary_from_buffers(offset): offset=offset) assert a[offset:] == b + with pytest.raises(TypeError, match="Argument 'type' has incorrect type"): + pa.DictionaryArray.from_buffers( + None, len(a), a.indices.buffers(), a.dictionary) + with pytest.raises(TypeError, match="Argument 'dictionary' has incorrect type"): + pa.DictionaryArray.from_buffers( + a.type, len(a), a.indices.buffers(), None) + @pytest.mark.numpy def test_dictionary_from_numpy(): @@ -4196,6 +4224,9 @@ def test_run_end_encoded_from_buffers(): pa.RunEndEncodedArray.from_buffers(ree_type, length, buffers, 1, offset, children) + with pytest.raises(TypeError, match="Argument 'type' has incorrect type"): + pa.RunEndEncodedArray.from_buffers(type=None, length=0, buffers=[]) + @pytest.mark.numpy def test_run_end_encoded_from_array_with_type(): diff --git a/python/pyarrow/tests/test_dataset.py b/python/pyarrow/tests/test_dataset.py index 0a94c0bd9875..49d6513b80fe 100644 --- a/python/pyarrow/tests/test_dataset.py +++ b/python/pyarrow/tests/test_dataset.py @@ -396,6 +396,12 @@ def test_filesystem_dataset(mockfs): # validation of required arguments with pytest.raises(TypeError, match="incorrect type"): ds.FileSystemDataset(fragments, file_format, schema) + with pytest.raises(TypeError, match="Fragment must not be None"): + ds.FileSystemDataset([None], schema=schema, format=file_format) + with pytest.raises(TypeError, match="Argument 'schema' has incorrect type"): + ds.FileSystemDataset(fragments, schema=None, format=file_format) + with pytest.raises(TypeError, match="Argument 'format' has incorrect type"): + ds.FileSystemDataset(fragments, schema=schema, format=None) # validation of root_partition with pytest.raises(TypeError, match="incorrect type"): ds.FileSystemDataset(fragments, schema=schema, diff --git a/python/pyarrow/tests/test_misc.py b/python/pyarrow/tests/test_misc.py index 856873f441b4..e7cc0a459e08 100644 --- a/python/pyarrow/tests/test_misc.py +++ b/python/pyarrow/tests/test_misc.py @@ -270,3 +270,32 @@ def test_extension_type_constructor_errors(klass): msg = f"Do not call {klass.__name__}'s constructor directly, use .* instead." with pytest.raises(TypeError, match=msg): klass() + + +@pytest.mark.processes +@pytest.mark.parametrize(("module", "call", "message"), [ + pytest.param( + "acero", "ScanNodeOptions(None)", "Argument 'dataset'", + marks=[pytest.mark.acero, pytest.mark.dataset], id="scan"), + pytest.param( + "acero", "ProjectNodeOptions([None])", "Expression must not be None", + marks=pytest.mark.acero, id="project"), + pytest.param( + "parquet", "ParquetSchema(None)", "Argument 'container'", + marks=pytest.mark.parquet, id="parquet-schema"), + pytest.param( + "parquet", "ColumnSchema(None, 0)", "Argument 'schema'", + marks=pytest.mark.parquet, id="column-schema"), + pytest.param( + "fs", 'SubTreeFileSystem("", None)', "Argument 'base_fs'", + id="subtree"), +]) +def test_required_arrow_objects_reject_none_without_crashing( + module: str, call: str, message: str +) -> None: + result = subprocess.run( + args=[sys.executable, "-c", f"import pyarrow.{module} as mod; mod.{call}"], + capture_output=True, text=True, timeout=30, + ) + assert result.returncode == 1, result.stderr + assert f"TypeError: {message}" in result.stderr