Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
97 changes: 91 additions & 6 deletions cpp/gdb_arrow.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@
# specific language governing permissions and limitations
# under the License.

from bisect import bisect_right
from collections import namedtuple
from collections.abc import Sequence
import datetime
Expand Down Expand Up @@ -95,7 +96,8 @@ def identity(v):


def has_null_bitmap(type_id):
return type_id not in (Type.NA, Type.SPARSE_UNION, Type.DENSE_UNION)
return type_id not in (Type.NA, Type.SPARSE_UNION, Type.DENSE_UNION,
Type.RUN_END_ENCODED)


@lru_cache()
Expand Down Expand Up @@ -626,11 +628,12 @@ def bytes_view(self, offset=0, length=None):
"""
Return a view over the bytes of this buffer.
"""
if self.size > 0:
if length is None:
length = self.size
if length is None:
length = self.size - offset
# Sliced arrays may share buffers, so only read the requested range.
if length > 0:
mem = gdb.selected_inferior().read_memory(
self.val['data_'] + offset, self.size)
self.val['data_'] + offset, length)
else:
mem = memoryview(b"")
# Read individual bytes as unsigned integers rather than
Expand Down Expand Up @@ -769,7 +772,8 @@ def __getitem__(self, index):
def from_buffer(cls, buf, offset, length):
assert isinstance(buf, Buffer)
byte_offset, bit_offset = divmod(offset, 8)
byte_length = math.ceil(length + offset / 8) - byte_offset
# E.g. offset=3, length=6 selects bits 3..8 and needs 2 bytes.
byte_length = math.ceil((bit_offset + length) / 8)
Comment on lines +775 to +776

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I see this fixes a bug, can you add a test that exercises it?

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sure. I will address these two comments later today.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The existing BooleanArray slice tests already cover this case:

auto heap_bool_array_sliced_1_9 =
    SliceArrayFromJSON(boolean(), json_bool_array, 1, 9);
auto heap_bool_array_sliced_2_6 =
    SliceArrayFromJSON(boolean(), json_bool_array, 2, 6);

They exercise Bitmap.from_buffer() with non-byte-aligned offsets and ranges that cross byte boundaries.

Before this change, Buffer.bytes_view() ignored the requested length:

if length is None:
    length = self.size
mem = gdb.selected_inferior().read_memory(
    self.val['data_'] + offset, self.size)

It always read self.size bytes. Now it reads the requested length:

mem = gdb.selected_inferior().read_memory(
    self.val['data_'] + offset, length)

This previously masked the incorrect byte-length calculation in Bitmap.from_buffer(). Since the existing tests already cover this path, I did not add another test case.

return cls(buf.bytes_view(byte_offset, byte_length),
bit_offset, length)

Expand Down Expand Up @@ -1069,6 +1073,7 @@ def num_rows(self):
'SparseUnionType': 'sparse_union',
'DenseUnionType': 'dense_union',
'DictionaryType': 'dictionary',
'RunEndEncodedType': 'run_end_encoded',
}


Expand Down Expand Up @@ -1170,6 +1175,20 @@ def to_string(self):
return f"{self._format_type()}({child})"


class RunEndEncodedTypePrinter(TypePrinter):
"""
Pretty-printer for run-end encoded types.
"""

def to_string(self):
fields = self.fields
if len(fields) != 2:
return f"{self._format_type()}<uninitialized or corrupt>"
run_end_type = fields[0].type
value_type = fields[1].type
return f"{self._format_type()}({run_end_type}, {value_type})"


class FixedSizeListTypePrinter(ListTypePrinter):
"""
Pretty-printer for fixed-size list type.
Expand Down Expand Up @@ -1465,6 +1484,18 @@ def to_string(self):
return f"{self._format_type()} of value {value}"


class RunEndEncodedScalarPrinter(ScalarPrinter):
"""
Pretty-printer for arrow::RunEndEncodedScalar.
"""

def to_string(self):
if not self.is_valid:
return self._format_null()
value = deref(self.val['value'])
return f"{self._format_type()} of value {value}"


class StructScalarPrinter(ScalarPrinter):
"""
Pretty-printer for arrow::StructScalar.
Expand Down Expand Up @@ -1831,6 +1862,51 @@ def children(self):
yield self._null_child(i)


class RunEndEncodedArrayDataPrinter(ArrayDataPrinter):
"""
ArrayDataPrinter specialization for run-end encoded arrays.
"""

def __init__(self, name, val):
if self.length == 0:
return
child_data = StdVector(self.val['child_data'])
self._run_ends_printer = ArrayDataPrinter(
"arrow::ArrayData", deref(child_data[0]))
self._values_printer = ArrayDataPrinter(
"arrow::ArrayData", deref(child_data[1]))

def display_hint(self):
return "array"

def children(self):
if self.length == 0:
return
run_ends = self._run_ends_printer._unpacked_buffer_values(
1, self._run_ends_printer.type_id)
values = iter(self._values_printer.children() or ())
run_index = bisect_right(run_ends, self.offset)
# Advance to the value for the run containing the logical offset.
for _ in range(run_index + 1):
value = next(values, None)
if value is None:
return

logical_index = self.offset
logical_end = self.offset + self.length
# Expand each run into the logical elements visible in this slice.
while logical_index < logical_end:
run_end = run_ends[run_index]
for i in range(logical_index, min(run_end, logical_end)):
yield self._valid_child(i - self.offset, value[1])
logical_index = run_end
run_index += 1
if logical_index < logical_end:
value = next(values, None)
if value is None:
return


class ArrayPrinter:
"""
Pretty-printer for arrow::Array and subclasses.
Expand Down Expand Up @@ -1999,6 +2075,13 @@ class FixedSizeListTypeClass(DataTypeClass):
scalar_printer = BaseListScalarPrinter


class RunEndEncodedTypeClass(DataTypeClass):
is_parametric = True
type_printer = RunEndEncodedTypePrinter
scalar_printer = RunEndEncodedScalarPrinter
array_data_printer = RunEndEncodedArrayDataPrinter

Comment thread
pitrou marked this conversation as resolved.

class MapTypeClass(DataTypeClass):
is_parametric = True
type_printer = MapTypePrinter
Expand Down Expand Up @@ -2093,6 +2176,8 @@ class ExtensionTypeClass(DataTypeClass):

Type.DICTIONARY: DataTypeTraits(DictionaryTypeClass, 'DictionaryType'),
Type.EXTENSION: DataTypeTraits(ExtensionTypeClass, 'ExtensionType'),
Type.RUN_END_ENCODED: DataTypeTraits(RunEndEncodedTypeClass,
'RunEndEncodedType'),
}


Expand Down
29 changes: 29 additions & 0 deletions python/pyarrow/src/arrow/python/gdb.cc
Original file line number Diff line number Diff line change
Expand Up @@ -222,6 +222,9 @@ void TestSession() {
FixedSizeListType fixed_size_list_type(float64(), 3);
auto heap_fixed_size_list_type = fixed_size_list(float64(), 3);

RunEndEncodedType run_end_encoded_type(int32(), utf8());
auto heap_run_end_encoded_type = run_end_encoded(int32(), utf8());

DictionaryType dict_type_unordered(int16(), utf8());
DictionaryType dict_type_ordered(int16(), utf8(), /*ordered=*/true);
auto heap_dict_type = dictionary(int16(), utf8());
Expand Down Expand Up @@ -389,6 +392,14 @@ void TestSession() {
FixedSizeListScalar fixed_size_list_scalar_null{
list_value_array, fixed_size_list(int32(), 3), /*is_valid=*/false};

auto run_end_encoded_scalar_type = run_end_encoded(int32(), utf8());
RunEndEncodedScalar run_end_encoded_scalar{MakeScalar("foo"),
run_end_encoded_scalar_type};
RunEndEncodedScalar run_end_encoded_scalar_null{run_end_encoded_scalar_type};
std::shared_ptr<Scalar> heap_run_end_encoded_scalar =
std::make_shared<RunEndEncodedScalar>(MakeScalar("foo"),
run_end_encoded_scalar_type);

auto struct_scalar_type = struct_({field("ints", int32()), field("strs", utf8())});
StructScalar struct_scalar{
ScalarVector{MakeScalar(int32_t(42)), MakeScalar("some text")}, struct_scalar_type};
Expand Down Expand Up @@ -448,6 +459,24 @@ void TestSession() {
auto heap_list_array = SliceArrayFromJSON(list(int64()), "[[1, 2], null, []]");
ListArray list_array{heap_list_array->data()};

// Encodes ["foo", "foo", null, null, null].
auto run_end_encoded_run_ends = SliceArrayFromJSON(int32(), "[2, 5]");
auto run_end_encoded_values = SliceArrayFromJSON(utf8(), R"(["foo", null])");
std::shared_ptr<Array> heap_run_end_encoded_array = *RunEndEncodedArray::Make(
/*logical_length=*/5, run_end_encoded_run_ends, run_end_encoded_values);
RunEndEncodedArray run_end_encoded_array{heap_run_end_encoded_array->data()};
auto heap_run_end_encoded_array_sliced = heap_run_end_encoded_array->Slice(1, 3);

// Sliced children
std::shared_ptr<Array> heap_ree_sliced = *RunEndEncodedArray::Make(
/*logical_length=*/3, SliceArrayFromJSON(int32(), "[1, 2, 5, 0]", 1, 2),
SliceArrayFromJSON(utf8(), R"(["bar", "baz", "foo", null, "qux"])", 2, 2),
/*logical_offset=*/1);

std::shared_ptr<Array> heap_ree_int64 = *RunEndEncodedArray::Make(
/*logical_length=*/3, SliceArrayFromJSON(int64(), "[1, 3]"),
SliceArrayFromJSON(int32(), "[42, null]"));

const char* json_double_array = "[-1.5, null]";
auto heap_double_array = SliceArrayFromJSON(float64(), json_double_array);

Expand Down
50 changes: 50 additions & 0 deletions python/pyarrow/tests/test_gdb.py
Original file line number Diff line number Diff line change
Expand Up @@ -407,6 +407,9 @@ def test_types_stack(gdb_arrow):
"arrow::large_list(arrow::large_utf8())")
check_stack_repr(gdb_arrow, "fixed_size_list_type",
"arrow::fixed_size_list(arrow::float64(), 3)")
check_stack_repr(
gdb_arrow, "run_end_encoded_type",
"arrow::run_end_encoded(arrow::int32(), arrow::utf8())")
check_stack_repr(
gdb_arrow, "map_type_unsorted",
"arrow::map(arrow::utf8(), arrow::binary(), keys_sorted=false)")
Expand Down Expand Up @@ -468,6 +471,9 @@ def test_types_heap(gdb_arrow):
"arrow::large_list(arrow::large_utf8())")
check_heap_repr(gdb_arrow, "heap_fixed_size_list_type",
"arrow::fixed_size_list(arrow::float64(), 3)")
check_heap_repr(
gdb_arrow, "heap_run_end_encoded_type",
"arrow::run_end_encoded(arrow::int32(), arrow::utf8())")
check_heap_repr(
gdb_arrow, "heap_map_type",
"arrow::map(arrow::utf8(), arrow::binary(), keys_sorted=false)")
Expand Down Expand Up @@ -744,6 +750,14 @@ def test_scalars_stack(gdb_arrow):
gdb_arrow, "fixed_size_list_scalar_null",
('arrow::FixedSizeListScalar of type '
'arrow::fixed_size_list(arrow::int32(), 3), null value'))
check_stack_repr(
gdb_arrow, "run_end_encoded_scalar",
('arrow::RunEndEncodedScalar of value '
'arrow::StringScalar of size 3, value "foo"'))
check_stack_repr(
gdb_arrow, "run_end_encoded_scalar_null",
('arrow::RunEndEncodedScalar of type '
'arrow::run_end_encoded(arrow::int32(), arrow::utf8()), null value'))

check_stack_repr(
gdb_arrow, "struct_scalar",
Expand Down Expand Up @@ -810,6 +824,10 @@ def test_scalars_heap(gdb_arrow):
gdb_arrow, "heap_map_scalar_null",
('arrow::MapScalar of type arrow::map(arrow::utf8(), arrow::int32(), '
'keys_sorted=false), null value'))
check_heap_repr(
gdb_arrow, "heap_run_end_encoded_scalar",
('arrow::RunEndEncodedScalar of value '
'arrow::StringScalar of size 3, value "foo"'))


def test_array_data(gdb_arrow):
Expand All @@ -828,6 +846,13 @@ def test_arrays_stack(gdb_arrow):
gdb_arrow, "list_array",
("arrow::ListArray of type arrow::list(arrow::int64()), "
"length 3, offset 0, null count 1"))
check_stack_repr(
gdb_arrow, "run_end_encoded_array",
("arrow::RunEndEncodedArray of type "
"arrow::run_end_encoded(arrow::int32(), arrow::utf8()), "
"length 5, offset 0, null count 0 = "
"{[0] = \"foo\", [1] = \"foo\", [2] = null, "
"[3] = null, [4] = null}"))


def test_arrays_heap(gdb_arrow):
Expand Down Expand Up @@ -1081,6 +1106,31 @@ def test_arrays_heap(gdb_arrow):
gdb_arrow, "heap_list_array",
("arrow::ListArray of type arrow::list(arrow::int64()), "
"length 3, offset 0, null count 1"))
check_heap_repr(
gdb_arrow, "heap_run_end_encoded_array",
("arrow::RunEndEncodedArray of type "
"arrow::run_end_encoded(arrow::int32(), arrow::utf8()), "
"length 5, offset 0, null count 0 = "
"{[0] = \"foo\", [1] = \"foo\", [2] = null, "
"[3] = null, [4] = null}"))
check_heap_repr(
gdb_arrow, "heap_run_end_encoded_array_sliced",
("arrow::RunEndEncodedArray of type "
"arrow::run_end_encoded(arrow::int32(), arrow::utf8()), "
"length 3, offset 1, null count 0 = "
"{[0] = \"foo\", [1] = null, [2] = null}"))
check_heap_repr(
gdb_arrow, "heap_ree_sliced",
("arrow::RunEndEncodedArray of type "
"arrow::run_end_encoded(arrow::int32(), arrow::utf8()), "
"length 3, offset 1, null count 0 = "
"{[0] = \"foo\", [1] = null, [2] = null}"))
check_heap_repr(
gdb_arrow, "heap_ree_int64",
("arrow::RunEndEncodedArray of type "
"arrow::run_end_encoded(arrow::int64(), arrow::int32()), "
"length 3, offset 0, null count 0 = "
"{[0] = 42, [1] = null, [2] = null}"))


def test_schema(gdb_arrow):
Expand Down
Loading