From cca44b5df69034f09041ea7180a8d8938801cad9 Mon Sep 17 00:00:00 2001 From: tadeja Date: Thu, 1 Oct 2026 23:19:18 +0200 Subject: [PATCH 1/4] GH-51638: [CI][C++][Parquet] Update expected JSON output in parquet-reader-test for simdjson 5.0 (#51661) ### Rationale for this change See #51638 - `fractured_json` formatting changes with the new simdjson version 5.0 (whitespace, line breaks, key order in tables), cause `parquet-reader-test` to fail on four of its tests `TestJSONWithLocalFile.JSONOutput...` ### What changes are included in this PR? Check `simdjson::SIMDJSON_VERSION_MAJOR`, and for `>= 5` use the new expected formatting (JSON content hasn't changed). Keep existing expected output formatting for simdjson 4.x ### Are these changes tested? Yes, CI runs pass. ### Are there any user-facing changes? No, only test changes. ### Was AI used for this PR? **PR code and description written by:** - [x] Human - [x] AI **Reviewed before submission by:** - [x] Human - [ ] AI - [ ] Not reviewed * GitHub Issue: #51638 Authored-by: Tadeja Kadunc Signed-off-by: Sutou Kouhei --- cpp/src/parquet/reader_test.cc | 228 ++++++++++++++++++++++++++++----- 1 file changed, 196 insertions(+), 32 deletions(-) diff --git a/cpp/src/parquet/reader_test.cc b/cpp/src/parquet/reader_test.cc index e4d2fd1d5900..b5487fa46899 100644 --- a/cpp/src/parquet/reader_test.cc +++ b/cpp/src/parquet/reader_test.cc @@ -1124,6 +1124,11 @@ class TestJSONWithLocalFile : public ::testing::Test { }; TEST_F(TestJSONWithLocalFile, JSONOutputWithStatistics) { + // simdjson 5.0 changed the format of fractured_json + if constexpr (simdjson::SIMDJSON_VERSION_MAJOR < 5) { + GTEST_SKIP() << "Test requires simdjson >= 5"; + } + std::string json_output = R"###({ "FileName": "nested_lists.snappy.parquet", "Version": "1.0", @@ -1138,15 +1143,9 @@ TEST_F(TestJSONWithLocalFile, JSONOutputWithStatistics) { "Name": "a.list.element.list.element.list.element", "PhysicalType": "BYTE_ARRAY", "ConvertedType": "UTF8", - "LogicalType": { "Type": "String" } + "LogicalType": {"Type": "String"} }, - { - "Id": "1", - "Name": "b", - "PhysicalType": "INT32", - "ConvertedType": "NONE", - "LogicalType": { "Type": "None" } - } + { "Id": "1", "Name": "b", "PhysicalType": "INT32", "ConvertedType": "NONE", "LogicalType": {"Type": "None"} } ], "RowGroups": [ { @@ -1191,6 +1190,11 @@ TEST_F(TestJSONWithLocalFile, JSONOutputWithStatistics) { } TEST_F(TestJSONWithLocalFile, JSONOutput) { + // simdjson 5.0 changed the format of fractured_json + if constexpr (simdjson::SIMDJSON_VERSION_MAJOR < 5) { + GTEST_SKIP() << "Test requires simdjson >= 5"; + } + std::string json_output = R"###({ "FileName": "alltypes_plain.parquet", "Version": "1.0", @@ -1200,17 +1204,77 @@ TEST_F(TestJSONWithLocalFile, JSONOutput) { "NumberOfRealColumns": "11", "NumberOfColumns": "11", "Columns": [ - { "ConvertedType": "NONE", "Id": "0" , "LogicalType": { "Type": "None" }, "Name": "id" , "PhysicalType": "INT32" }, - { "ConvertedType": "NONE", "Id": "1" , "LogicalType": { "Type": "None" }, "Name": "bool_col" , "PhysicalType": "BOOLEAN" }, - { "ConvertedType": "NONE", "Id": "2" , "LogicalType": { "Type": "None" }, "Name": "tinyint_col" , "PhysicalType": "INT32" }, - { "ConvertedType": "NONE", "Id": "3" , "LogicalType": { "Type": "None" }, "Name": "smallint_col" , "PhysicalType": "INT32" }, - { "ConvertedType": "NONE", "Id": "4" , "LogicalType": { "Type": "None" }, "Name": "int_col" , "PhysicalType": "INT32" }, - { "ConvertedType": "NONE", "Id": "5" , "LogicalType": { "Type": "None" }, "Name": "bigint_col" , "PhysicalType": "INT64" }, - { "ConvertedType": "NONE", "Id": "6" , "LogicalType": { "Type": "None" }, "Name": "float_col" , "PhysicalType": "FLOAT" }, - { "ConvertedType": "NONE", "Id": "7" , "LogicalType": { "Type": "None" }, "Name": "double_col" , "PhysicalType": "DOUBLE" }, - { "ConvertedType": "NONE", "Id": "8" , "LogicalType": { "Type": "None" }, "Name": "date_string_col", "PhysicalType": "BYTE_ARRAY" }, - { "ConvertedType": "NONE", "Id": "9" , "LogicalType": { "Type": "None" }, "Name": "string_col" , "PhysicalType": "BYTE_ARRAY" }, - { "ConvertedType": "NONE", "Id": "10", "LogicalType": { "Type": "None" }, "Name": "timestamp_col" , "PhysicalType": "INT96" } + { "Id": "0", "Name": "id", "PhysicalType": "INT32", "ConvertedType": "NONE", "LogicalType": {"Type": "None"} }, + { + "Id": "1", + "Name": "bool_col", + "PhysicalType": "BOOLEAN", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "2", + "Name": "tinyint_col", + "PhysicalType": "INT32", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "3", + "Name": "smallint_col", + "PhysicalType": "INT32", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "4", + "Name": "int_col", + "PhysicalType": "INT32", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "5", + "Name": "bigint_col", + "PhysicalType": "INT64", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "6", + "Name": "float_col", + "PhysicalType": "FLOAT", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "7", + "Name": "double_col", + "PhysicalType": "DOUBLE", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "8", + "Name": "date_string_col", + "PhysicalType": "BYTE_ARRAY", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "9", + "Name": "string_col", + "PhysicalType": "BYTE_ARRAY", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "10", + "Name": "timestamp_col", + "PhysicalType": "INT96", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + } ], "RowGroups": [ { @@ -1219,17 +1283,105 @@ TEST_F(TestJSONWithLocalFile, JSONOutput) { "TotalCompressedBytes": "0", "Rows": "8", "ColumnChunks": [ - { "CompressedSize": "73" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "0" , "StatsSet": "False", "UncompressedSize": "73" , "Values": "8" }, - { "CompressedSize": "24" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "1" , "StatsSet": "False", "UncompressedSize": "24" , "Values": "8" }, - { "CompressedSize": "47" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "2" , "StatsSet": "False", "UncompressedSize": "47" , "Values": "8" }, - { "CompressedSize": "47" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "3" , "StatsSet": "False", "UncompressedSize": "47" , "Values": "8" }, - { "CompressedSize": "47" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "4" , "StatsSet": "False", "UncompressedSize": "47" , "Values": "8" }, - { "CompressedSize": "55" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "5" , "StatsSet": "False", "UncompressedSize": "55" , "Values": "8" }, - { "CompressedSize": "47" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "6" , "StatsSet": "False", "UncompressedSize": "47" , "Values": "8" }, - { "CompressedSize": "55" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "7" , "StatsSet": "False", "UncompressedSize": "55" , "Values": "8" }, - { "CompressedSize": "88" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "8" , "StatsSet": "False", "UncompressedSize": "88" , "Values": "8" }, - { "CompressedSize": "49" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "9" , "StatsSet": "False", "UncompressedSize": "49" , "Values": "8" }, - { "CompressedSize": "139", "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "10", "StatsSet": "False", "UncompressedSize": "139", "Values": "8" } + { + "Id": "0", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "73", + "CompressedSize": "73" + }, + { + "Id": "1", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "24", + "CompressedSize": "24" + }, + { + "Id": "2", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "47", + "CompressedSize": "47" + }, + { + "Id": "3", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "47", + "CompressedSize": "47" + }, + { + "Id": "4", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "47", + "CompressedSize": "47" + }, + { + "Id": "5", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "55", + "CompressedSize": "55" + }, + { + "Id": "6", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "47", + "CompressedSize": "47" + }, + { + "Id": "7", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "55", + "CompressedSize": "55" + }, + { + "Id": "8", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "88", + "CompressedSize": "88" + }, + { + "Id": "9", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "49", + "CompressedSize": "49" + }, + { + "Id": "10", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "139", + "CompressedSize": "139" + } ] } ] @@ -1243,6 +1395,11 @@ TEST_F(TestJSONWithLocalFile, JSONOutput) { TEST_F(TestJSONWithLocalFile, JSONOutputFLBA) { // min-max stats for FLBA contains non-utf8 output, so we don't check // the whole json output. + // simdjson 5.0 changed the format of fractured_json + if constexpr (simdjson::SIMDJSON_VERSION_MAJOR < 5) { + GTEST_SKIP() << "Test requires simdjson >= 5"; + } + std::string json_content = ReadFromLocalFile("fixed_length_byte_array.parquet"); std::string json_contains = R"###({ @@ -1259,7 +1416,7 @@ TEST_F(TestJSONWithLocalFile, JSONOutputFLBA) { "Name": "flba_field", "PhysicalType": "FIXED_LEN_BYTE_ARRAY(4)", "ConvertedType": "NONE", - "LogicalType": { "Type": "None" } + "LogicalType": {"Type": "None"} } ],)###"; @@ -1267,11 +1424,18 @@ TEST_F(TestJSONWithLocalFile, JSONOutputFLBA) { } TEST_F(TestJSONWithLocalFile, JSONOutputSortColumns) { + // simdjson 5.0 changed the format of fractured_json + if constexpr (simdjson::SIMDJSON_VERSION_MAJOR < 5) { + GTEST_SKIP() << "Test requires simdjson >= 5"; + } + std::string json_content = ReadFromLocalFile("sort_columns.parquet"); std::string json_contains = R"###("SortColumns": [ - { "column_idx": 0, "descending": 1, "nulls_first": 1 }, { "column_idx": 1, "descending": 0, "nulls_first": 0 } + {"column_idx": 0, "descending": 1, "nulls_first": 1}, + {"column_idx": 1, "descending": 0, "nulls_first": 0} ],)###"; + EXPECT_THAT(json_content, testing::HasSubstr(json_contains)); } From 1deceed534441193b79006331d5ef2a6b9dda99d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ra=C3=BAl=20Cumplido?= Date: Thu, 1 Oct 2026 23:44:58 +0200 Subject: [PATCH 2/4] GH-51675: [Python][Packaging] Apply upstream fix to UriParser from vcpkg and force Windows image rebuild to fix builds on wheels (#51676) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Rationale for this change We have bumped UriParser from a vendored version to be built from external sources. We have to rebuild Windows wheels and we have to apply a fix for older GLIBC see the following upstream commit: - https://github.com/uriparser/uriparser/commit/072a97c604a05d120686d8ada8eff093a983d75d ### What changes are included in this PR? Apply to vcpkg the upstream commit to fix older GLIBC. ### Are these changes tested? Yes via archery wheels builds ### Are there any user-facing changes? No ### Was AI used for this PR? In accordance to the [AI generation guidelines](https://arrow.apache.org/docs/dev/developers/overview.html#ai-generated-code), please disclose below whether and how AI was used in this PR. **PR code and description written by:** - [x] Human - [x] AI I've used AI to analyze the errors. Changes have been added and applied by me locally. **Reviewed before submission by:** - [x] Human - [ ] AI - [ ] Not reviewed * GitHub Issue: #51675 Authored-by: Raúl Cumplido Signed-off-by: Sutou Kouhei --- .env | 2 +- ci/vcpkg/ports.patch | 34 ++++++++++++++++++++++++++++++++++ 2 files changed, 35 insertions(+), 1 deletion(-) diff --git a/.env b/.env index f0d538ccd934..cd30f7aa7c8c 100644 --- a/.env +++ b/.env @@ -98,6 +98,6 @@ VCPKG="9b965a116838c6cdcd36bca60d1b81b030c8ab8d" # 2026.05.27 (not release, u # ci/docker/python-*-windows-*.dockerfile or the vcpkg config. # This is a workaround for our CI problem that "archery docker build" doesn't # use pulled built images in dev/tasks/python-wheels/github.windows.yml. -PYTHON_WHEEL_WINDOWS_IMAGE_REVISION=2026-09-09 +PYTHON_WHEEL_WINDOWS_IMAGE_REVISION=2026-10-01 PYTHON_WHEEL_WINDOWS_TEST_IMAGE_REVISION=2026-09-14 diff --git a/ci/vcpkg/ports.patch b/ci/vcpkg/ports.patch index 8e62c75e7ac7..db38c25de3f8 100644 --- a/ci/vcpkg/ports.patch +++ b/ci/vcpkg/ports.patch @@ -80,3 +80,37 @@ index 278bc17a1c..d47d859360 100644 ) file(GLOB modules "${SOURCE_PATH}/cmake_modules/Find*.cmake") file(REMOVE ${modules} "${SOURCE_PATH}/c++/libs/libhdfspp/libhdfspp.tar.gz") +diff --git a/ports/uriparser/fix-glibc-2.28-reallocarray.diff b/ports/uriparser/fix-glibc-2.28-reallocarray.diff +new file mode 100644 +index 0000000..57adef4 +--- /dev/null ++++ b/ports/uriparser/fix-glibc-2.28-reallocarray.diff +@@ -0,0 +1,15 @@ ++diff --git a/src/UriMemory.c b/src/UriMemory.c ++--- a/src/UriMemory.c +++++ b/src/UriMemory.c ++@@ -50,6 +50,11 @@ ++ # define _DEFAULT_SOURCE 1 ++ # endif ++ +++// For glibc <2.29 +++# if !defined(_GNU_SOURCE) +++# define _GNU_SOURCE 1 +++# endif +++ ++ // For NetBSD (stdlib.h revision 1.122 of 2020-05-26) ++ # if defined(__NetBSD__) && !defined(_OPENBSD_SOURCE) ++ # define _OPENBSD_SOURCE 1 +diff --git a/ports/uriparser/portfile.cmake b/ports/uriparser/portfile.cmake +index 5df5e94..225e5fb 100644 +--- a/ports/uriparser/portfile.cmake ++++ b/ports/uriparser/portfile.cmake +@@ -4,6 +4,8 @@ vcpkg_from_github( + REF "uriparser-${VERSION}" + SHA512 1e4c397418e71e705b5712de2dc3ea6e95163c9d95bfbaa5f7d2fd2344d34eaf167d7d9389ab9671cfb314af026621b56e632ea34b681c2f4d7951a1b32a98b4 + HEAD_REF master ++ PATCHES ++ fix-glibc-2.28-reallocarray.diff + ) + + if("tool" IN_LIST FEATURES) From 12409f87ac47dc0a20da11f27d10b07cd686a8b4 Mon Sep 17 00:00:00 2001 From: Hiroyuki Sato Date: Fri, 2 Oct 2026 17:46:55 +0900 Subject: [PATCH 3/4] GH-51682: [CI][C++] Add liburiparser-dev to system dependency image (#51683) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Rationale for this change Changes around UriParser, including #51007, caused example-cpp-minimal-build-static-system-dependency to fail to build. ### What changes are included in this PR? This change adds liburiparser-dev to the system dependency image. ### Are these changes tested? Yes. ### Are there any user-facing changes? No. ### Was AI used for this PR? In accordance to the [AI generation guidelines](https://arrow.apache.org/docs/dev/developers/overview.html#ai-generated-code), please disclose below whether and how AI was used in this PR. **PR code and description written by:** - [x] Human - [ ] AI **Reviewed before submission by:** - [x] Human - [ ] AI - [ ] Not reviewed * GitHub Issue: #51682 Authored-by: Hiroyuki Sato Signed-off-by: Raúl Cumplido --- cpp/examples/minimal_build/system_dependency.dockerfile | 1 + 1 file changed, 1 insertion(+) diff --git a/cpp/examples/minimal_build/system_dependency.dockerfile b/cpp/examples/minimal_build/system_dependency.dockerfile index 773cd52baab4..ac7cef3496ea 100644 --- a/cpp/examples/minimal_build/system_dependency.dockerfile +++ b/cpp/examples/minimal_build/system_dependency.dockerfile @@ -34,6 +34,7 @@ RUN apt-get update -y -q && \ libre2-dev \ libsnappy-dev \ libthrift-dev \ + liburiparser-dev \ libutf8proc-dev \ libzstd-dev \ pkg-config \ From 28e6c675fcb56c4d2f2fe64b612aa09b49f7b811 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ra=C3=BAl=20Cumplido?= Date: Fri, 2 Oct 2026 15:16:47 +0200 Subject: [PATCH 4/4] GH-51687: [C++] Skipt tests that require threads if ARROW_ENABLE_THREADING=OFF (#51688) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Rationale for this change The following PR introduced some tests that require threeading but those are not skipped when ARROW_ENABLE_THREADING=OFF failing on jobs like emscripten. - https://github.com/apache/arrow/pull/51498 ### What changes are included in this PR? Skip the tests if ARROW_ENABLE_THREADING=OFF ### Are these changes tested? Yes via archer with the failing job. ### Are there any user-facing changes? No ### Was AI used for this PR? In accordance to the [AI generation guidelines](https://arrow.apache.org/docs/dev/developers/overview.html#ai-generated-code), please disclose below whether and how AI was used in this PR. **PR code and description written by:** - [x] Human - [ ] AI **Reviewed before submission by:** - [x] Human - [ ] AI - [ ] Not reviewed * GitHub Issue: #51687 Authored-by: Raúl Cumplido Signed-off-by: Rossi Sun --- cpp/src/arrow/util/async_generator_test.cc | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/cpp/src/arrow/util/async_generator_test.cc b/cpp/src/arrow/util/async_generator_test.cc index 9d965dbc4e5f..e51410b73e81 100644 --- a/cpp/src/arrow/util/async_generator_test.cc +++ b/cpp/src/arrow/util/async_generator_test.cc @@ -976,6 +976,9 @@ TEST_F(MergedGeneratorErrorHookTest, OuterErrorToWaiterNotOvertakenByLaterPull) // receives the error has completed, even when it is requested while the generator is // already completing. TEST_F(MergedGeneratorErrorHookTest, InnerErrorToWaiterNotOvertakenDuringCompletion) { +#ifndef ARROW_ENABLE_THREADING + GTEST_SKIP() << "Test requires threading support"; +#endif auto failing = Future::Make(); AsyncGenerator failing_sub = [failing]() { return failing; }; std::vector> subs = {failing_sub}; @@ -994,6 +997,9 @@ TEST_F(MergedGeneratorErrorHookTest, InnerErrorToWaiterNotOvertakenDuringComplet } TEST_F(MergedGeneratorErrorHookTest, OuterErrorToWaiterNotOvertakenDuringCompletion) { +#ifndef ARROW_ENABLE_THREADING + GTEST_SKIP() << "Test requires threading support"; +#endif auto failing = Future>::Make(); AsyncGenerator> source = [failing]() { return failing; }; MergedGenerator gen(std::move(source), 1); @@ -1011,6 +1017,9 @@ TEST_F(MergedGeneratorErrorHookTest, OuterErrorToWaiterNotOvertakenDuringComplet } TEST_F(MergedGeneratorErrorHookTest, ClaimedErrorNotOvertakenDuringCompletion) { +#ifndef ARROW_ENABLE_THREADING + GTEST_SKIP() << "Test requires threading support"; +#endif auto failing = Future::Make(); AsyncGenerator failing_sub = [failing]() { return failing; }; std::vector> subs = {MakeVectorGenerator({TestInt(1)}),