From d9f14845d4026bb88a3241ca85e2b3bb5ba67f11 Mon Sep 17 00:00:00 2001 From: Edmond Abraham <54119952+gitedmond@users.noreply.github.com> Date: Mon, 5 Oct 2026 15:09:10 -0400 Subject: [PATCH] GH-51760: [C++][Python] Emit decimal points for integral floats --- .../arrow/compute/kernels/scalar_cast_test.cc | 9 ++++-- cpp/src/arrow/pretty_print_test.cc | 24 +++++++-------- cpp/src/arrow/scalar_test.cc | 3 +- cpp/src/arrow/util/formatting.cc | 6 ++-- cpp/src/arrow/util/formatting_util_test.cc | 30 +++++++++++-------- python/pyarrow/tests/test_dataset.py | 19 ++++++++++++ 6 files changed, 61 insertions(+), 30 deletions(-) diff --git a/cpp/src/arrow/compute/kernels/scalar_cast_test.cc b/cpp/src/arrow/compute/kernels/scalar_cast_test.cc index 864ec4afce0c..4cfbaec34d2e 100644 --- a/cpp/src/arrow/compute/kernels/scalar_cast_test.cc +++ b/cpp/src/arrow/compute/kernels/scalar_cast_test.cc @@ -3571,9 +3571,12 @@ TEST(Cast, IntToString) { TEST(Cast, FloatingToString) { for (auto float_type : {float16(), float32(), float64()}) { for (auto string_type : {utf8(), utf8_view(), large_utf8()}) { - CheckCast(ArrayFromJSON(float_type, "[0.0, -0.0, 1.5, -Inf, Inf, NaN, null]"), - ArrayFromJSON(string_type, - R"(["0", "-0", "1.5", "-inf", "inf", "nan", null])")); + CheckCast( + ArrayFromJSON(float_type, + "[0.0, -0.0, 20.0, -20.0, 1.5, -Inf, Inf, NaN, null]"), + ArrayFromJSON( + string_type, + R"(["0.0", "-0.0", "20.0", "-20.0", "1.5", "-inf", "inf", "nan", null])")); } } } diff --git a/cpp/src/arrow/pretty_print_test.cc b/cpp/src/arrow/pretty_print_test.cc index c90b03bbda38..73281e77cbc4 100644 --- a/cpp/src/arrow/pretty_print_test.cc +++ b/cpp/src/arrow/pretty_print_test.cc @@ -147,18 +147,18 @@ TEST_F(TestPrettyPrint, PrimitiveType) { std::vector values2 = {0., 1., 2., 3., 4.}; static const char* ex2 = R"expected([ - 0, - 1, + 0.0, + 1.0, null, - 3, + 3.0, null ])expected"; CheckPrimitive({0, 10}, is_valid, values2, ex2); static const char* ex2_in2 = R"expected( [ - 0, - 1, + 0.0, + 1.0, null, - 3, + 3.0, null ])expected"; CheckPrimitive({2, 10}, is_valid, values2, ex2_in2); @@ -337,16 +337,16 @@ TEST_F(TestPrettyPrint, UInt64) { TEST_F(TestPrettyPrint, HalfFloat) { static const char* expected = R"expected([ -inf, - -1234, - -0, - 0, - 1, + -1234.0, + -0.0, + 0.0, + 1.0, 1.2001953125, 2.5, 3.9921875, 4.125, - 10000, - 12344, + 10000.0, + 12344.0, inf, nan, null diff --git a/cpp/src/arrow/scalar_test.cc b/cpp/src/arrow/scalar_test.cc index 9594e8883b34..8c97843476b1 100644 --- a/cpp/src/arrow/scalar_test.cc +++ b/cpp/src/arrow/scalar_test.cc @@ -150,7 +150,8 @@ TEST(TestBooleanScalar, Cast) { ARROW_SCOPED_TRACE("to type: ", to_type->ToString()); ASSERT_OK_AND_ASSIGN(auto casted, Cast(scalar, to_type)); ASSERT_EQ(casted.scalar()->type->id(), to_type->id()); - ASSERT_EQ(casted.scalar()->ToString(), std::to_string(b)); + ASSERT_EQ(casted.scalar()->ToString(), + std::to_string(b) + (is_floating(*to_type) ? ".0" : "")); } // String type. diff --git a/cpp/src/arrow/util/formatting.cc b/cpp/src/arrow/util/formatting.cc index 58dadd0b11e5..11a11271055f 100644 --- a/cpp/src/arrow/util/formatting.cc +++ b/cpp/src/arrow/util/formatting.cc @@ -42,8 +42,10 @@ const char digit_pairs[] = struct FloatToStringFormatter::Impl { Impl() - : converter_(DoubleToStringConverter::EMIT_POSITIVE_EXPONENT_SIGN, "inf", "nan", - 'e', -6, 10, 6, 0) {} + : converter_(DoubleToStringConverter::EMIT_POSITIVE_EXPONENT_SIGN | + DoubleToStringConverter::EMIT_TRAILING_DECIMAL_POINT | + DoubleToStringConverter::EMIT_TRAILING_ZERO_AFTER_POINT, + "inf", "nan", 'e', -6, 10, 6, 0) {} Impl(int flags, const char* inf_symbol, const char* nan_symbol, char exp_character, int decimal_in_shortest_low, int decimal_in_shortest_high, diff --git a/cpp/src/arrow/util/formatting_util_test.cc b/cpp/src/arrow/util/formatting_util_test.cc index 186e5a8b4331..5a31f590ec97 100644 --- a/cpp/src/arrow/util/formatting_util_test.cc +++ b/cpp/src/arrow/util/formatting_util_test.cc @@ -246,12 +246,14 @@ TEST(Formatting, Int64) { TEST(Formatting, Float) { StringFormatter formatter; - AssertFormatting(formatter, 0.0f, "0"); - AssertFormatting(formatter, -0.0f, "-0"); + AssertFormatting(formatter, 0.0f, "0.0"); + AssertFormatting(formatter, -0.0f, "-0.0"); + AssertFormatting(formatter, 20.0f, "20.0"); + AssertFormatting(formatter, -20.0f, "-20.0"); AssertFormatting(formatter, 1.5f, "1.5"); AssertFormatting(formatter, 0.0001f, "0.0001"); AssertFormatting(formatter, 1234.567f, "1234.567"); - AssertFormatting(formatter, 1e9f, "1000000000"); + AssertFormatting(formatter, 1e9f, "1000000000.0"); AssertFormatting(formatter, 1e10f, "1e+10"); AssertFormatting(formatter, 1e20f, "1e+20"); AssertFormatting(formatter, 1e-6f, "0.000001"); @@ -266,12 +268,14 @@ TEST(Formatting, Float) { TEST(Formatting, Double) { StringFormatter formatter; - AssertFormatting(formatter, 0.0, "0"); - AssertFormatting(formatter, -0.0, "-0"); + AssertFormatting(formatter, 0.0, "0.0"); + AssertFormatting(formatter, -0.0, "-0.0"); + AssertFormatting(formatter, 20.0, "20.0"); + AssertFormatting(formatter, -20.0, "-20.0"); AssertFormatting(formatter, 1.5, "1.5"); AssertFormatting(formatter, 0.0001, "0.0001"); AssertFormatting(formatter, 1234.567, "1234.567"); - AssertFormatting(formatter, 1e9, "1000000000"); + AssertFormatting(formatter, 1e9, "1000000000.0"); AssertFormatting(formatter, 1e10, "1e+10"); AssertFormatting(formatter, 1e20, "1e+20"); AssertFormatting(formatter, 1e-6, "0.000001"); @@ -286,21 +290,23 @@ TEST(Formatting, Double) { TEST(Formatting, HalfFloat) { StringFormatter formatter; - AssertFormatting(formatter, Float16(0.0f).bits(), "0"); - AssertFormatting(formatter, Float16(-0.0f).bits(), "-0"); + AssertFormatting(formatter, Float16(0.0f).bits(), "0.0"); + AssertFormatting(formatter, Float16(-0.0f).bits(), "-0.0"); + AssertFormatting(formatter, Float16(20.0f).bits(), "20.0"); + AssertFormatting(formatter, Float16(-20.0f).bits(), "-20.0"); AssertFormatting(formatter, Float16(1.5f).bits(), "1.5"); // Slightly adapted from values present here // https://blogs.mathworks.com/cleve/2017/05/08/half-precision-16-bit-floating-point-arithmetic/ - AssertFormatting(formatter, 0x3c00, "1"); + AssertFormatting(formatter, 0x3c00, "1.0"); AssertFormatting(formatter, 0x3c01, "1.0009765625"); AssertFormatting(formatter, 0x0400, "0.00006103515625"); AssertFormatting(formatter, 0x0001, "5.960464477539063e-8"); // Can't avoid loss of precision here. - AssertFormatting(formatter, Float16(1234.567f).bits(), "1235"); - AssertFormatting(formatter, Float16(1e3f).bits(), "1000"); - AssertFormatting(formatter, Float16(1e4f).bits(), "10000"); + AssertFormatting(formatter, Float16(1234.567f).bits(), "1235.0"); + AssertFormatting(formatter, Float16(1e3f).bits(), "1000.0"); + AssertFormatting(formatter, Float16(1e4f).bits(), "10000.0"); AssertFormatting(formatter, Float16(1e10f).bits(), "inf"); AssertFormatting(formatter, Float16(1e15f).bits(), "inf"); diff --git a/python/pyarrow/tests/test_dataset.py b/python/pyarrow/tests/test_dataset.py index 0a94c0bd9875..d61a2be6aea5 100644 --- a/python/pyarrow/tests/test_dataset.py +++ b/python/pyarrow/tests/test_dataset.py @@ -3503,6 +3503,25 @@ def test_csv_format(tempdir, dataset_reader): assert result.equals(table) +@pytest.mark.parametrize("float_type", [pa.float32(), pa.float64()]) +def test_csv_integral_floats_preserve_inferred_type( + tempdir, float_type, dataset_reader): + tables = [ + pa.table({'x': pa.array([20.0, 21.0], type=float_type), 'i': [20, 21]}), + pa.table({'x': pa.array([20.5, 21.25], type=float_type), 'i': [22, 23]}), + ] + paths = [str(tempdir / f't{i}.csv') for i in range(len(tables))] + for table, path in zip(tables, paths): + pyarrow.csv.write_csv(table, path) + + assert pathlib.Path(paths[0]).read_text() == '"x","i"\n20.0,20\n21.0,21\n' + expected = pa.table({'x': [20.0, 21.0, 20.5, 21.25], + 'i': [20, 21, 22, 23]}) + dataset = ds.dataset(paths, format='csv') + assert dataset.schema == expected.schema + assert dataset_reader.to_table(dataset).equals(expected) + + @pytest.mark.pandas @pytest.mark.parametrize("compression", [ "bz2",