diff --git a/.env b/.env index f0d538ccd934..cd30f7aa7c8c 100644 --- a/.env +++ b/.env @@ -98,6 +98,6 @@ VCPKG="9b965a116838c6cdcd36bca60d1b81b030c8ab8d" # 2026.05.27 (not release, u # ci/docker/python-*-windows-*.dockerfile or the vcpkg config. # This is a workaround for our CI problem that "archery docker build" doesn't # use pulled built images in dev/tasks/python-wheels/github.windows.yml. -PYTHON_WHEEL_WINDOWS_IMAGE_REVISION=2026-09-09 +PYTHON_WHEEL_WINDOWS_IMAGE_REVISION=2026-10-01 PYTHON_WHEEL_WINDOWS_TEST_IMAGE_REVISION=2026-09-14 diff --git a/CHANGELOG.md b/CHANGELOG.md index 47a651185ad6..f44ea03e23d2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,4 +1,354 @@ +# Apache Arrow 26.0.0 (2026-10-01 00:00:00+00:00) + +## Bug Fixes + +* [GH-33432](https://github.com/apache/arrow/issues/33432) - [R] Match base/stringr semantics for str_replace() with NA replacement (#51197) +* [GH-34860](https://github.com/apache/arrow/issues/34860) - [R] New column name wrongly set when using mutate with if_any (#51314) +* [GH-35692](https://github.com/apache/arrow/issues/35692) - [C++][Parquet] Support to read fixed size list array with nulls (#50271) +* [GH-37004](https://github.com/apache/arrow/issues/37004) - [C++][Python] Fix dropped child data when viewing/casting e… (#50502) +* [GH-37476](https://github.com/apache/arrow/issues/37476) - [C++][Python] Preserve unsigned dictionary index types when building from values (#50475) +* [GH-37761](https://github.com/apache/arrow/issues/37761) - [R] Argument names ignored in schema supplied as in_type argument to register_scalar_function() (#51324) +* [GH-38358](https://github.com/apache/arrow/issues/38358) - [R][day] ) (#51292) +* [GH-39688](https://github.com/apache/arrow/issues/39688) - [R] "Error: Filter expression not supported for Arrow Datasets" using "date" expression rigth hand side of a filter (#51291) +* [GH-39961](https://github.com/apache/arrow/issues/39961) - [C++][Python] Propagate CSV parse delimiter to write options (#49858) +* [GH-40163](https://github.com/apache/arrow/issues/40163) - [Archery] Avoid setuptools_scm internal API (#50669) +* [GH-40303](https://github.com/apache/arrow/issues/40303) - [R] : Unnamed columns cause issues when used in dplyr queries (#51313) +* [GH-44183](https://github.com/apache/arrow/issues/44183) - [C++] Support run-end encoded struct, list (view), large list (view) and map values (#50534) +* [GH-45086](https://github.com/apache/arrow/issues/45086) - [C++] Fix heap buffer overflow in FillNullForward/Backward … (#50843) +* [GH-45373](https://github.com/apache/arrow/issues/45373) - [R] summarize after arrange fails (#51312) +* [GH-46454](https://github.com/apache/arrow/issues/46454) - [C++][Dataset][Acero] Preserve order when writting with TeeNode (#46455) +* [GH-46646](https://github.com/apache/arrow/issues/46646) - [dev][R] Replace linr with jarl for R linting / pre-commit check (#50851) +* [GH-47369](https://github.com/apache/arrow/issues/47369) - [Python][Parquet] Fix wrong variable in the invalid operator error message (#51004) +* [GH-48072](https://github.com/apache/arrow/issues/48072) - [C++][Acero] fix a bug of materialize boolean (#48073) +* [GH-48137](https://github.com/apache/arrow/issues/48137) - [C++] Restore ThreadPool state when a worker fails to start (#51107) +* [GH-48344](https://github.com/apache/arrow/issues/48344) - [Python] Fix Table.from_struct_array for empty ChunkedArray (#49869) +* [GH-48679](https://github.com/apache/arrow/issues/48679) - [C++] Fix pivot_wider with non-monotonic group ids (#50423) +* [GH-48977](https://github.com/apache/arrow/issues/48977) - [C++] Fix quadratic field name index construction on libc++ (#50970) +* [GH-49482](https://github.com/apache/arrow/issues/49482) - [C++][FlightRPC][ODBC] Fix inconsistent SQLGetInfo values in global connection (#50021) +* [GH-49511](https://github.com/apache/arrow/issues/49511) - Fix rare doctest Table.join_asof and default_memory_pool failures (#50446) +* [GH-49826](https://github.com/apache/arrow/issues/49826) - [Python] Return NotImplemented from Scalar/Array arithmetic dunders for unsupported types (#49845) +* [GH-49889](https://github.com/apache/arrow/issues/49889) - [C++][Compute] Handle logical nulls in validity kernels (#50269) +* [GH-50140](https://github.com/apache/arrow/issues/50140) - [C++][Gandiva] Fix castVARCHAR(decimal128) native memory corruption / SIGSEGV on allocation failure (#50141) +* [GH-50148](https://github.com/apache/arrow/issues/50148) - [C++] Add Content-Encoding support to S3 filesystem metadata (#50167) +* [GH-50186](https://github.com/apache/arrow/issues/50186) - [C++][Gandiva] REPLACE throws "Buffer overflow for output string" for results larger than 64 KB (#50187) +* [GH-50222](https://github.com/apache/arrow/issues/50222) - [C++] Use FetchContent for xsimd (#50303) +* [GH-50239](https://github.com/apache/arrow/issues/50239) - [R] Data race issue from R API requests in parallel region (#50488) +* [GH-50302](https://github.com/apache/arrow/issues/50302) - [GLib][Ruby][FlightRPC] Fix GC related problems (#50401) +* [GH-50311](https://github.com/apache/arrow/issues/50311) - [C++] `KeyValueMetadata::Delete` returns IndexError instead of crashing due to seg fault (#50322) +* [GH-50312](https://github.com/apache/arrow/issues/50312) - [Python] Fix UUID extension type round-trip to pandas returning bytes (#50325) +* [GH-50316](https://github.com/apache/arrow/issues/50316) - [C++][CI] Install libboost-process-dev on Debian experimental (#50323) +* [GH-50339](https://github.com/apache/arrow/issues/50339) - [R] read_ipc_stream fails to unify nested uint64 fields inside a Struct array across record batches (#50374) +* [GH-50358](https://github.com/apache/arrow/issues/50358) - [Release] Fix permission issues when binary signing release candidate artifacts (#50359) +* [GH-50360](https://github.com/apache/arrow/issues/50360) - [Release] Remove stray Apache-Arrow-Flight-SQL-ODBC-*-win64.msi from 04-binary-download.sh and 05-binary-upload.sh (#50362) +* [GH-50378](https://github.com/apache/arrow/issues/50378) - [R] Reading a parquet with a Float16 column yields incorrect value (#50451) +* [GH-50388](https://github.com/apache/arrow/issues/50388) - [C++][CI] Make ccache effective MSVC-based builds (#50387) +* [GH-50399](https://github.com/apache/arrow/issues/50399) - [C++][Gandiva] Use timegm in DaysSince helper in Gandiva date_time_test (#50400) +* [GH-50403](https://github.com/apache/arrow/issues/50403) - [C++] Remove ArrayBuilder::AppendToBitmap (#50404) +* [GH-50424](https://github.com/apache/arrow/issues/50424) - [CI] install Chrome latest for test-conda-python-emscripten (#50425) +* [GH-50476](https://github.com/apache/arrow/issues/50476) - [R] empty_named_list() uses deprecated .Names in structure() (#50485) +* [GH-50481](https://github.com/apache/arrow/issues/50481) - [C++] Fix CSV reader mis-parsing rows with an embedded NUL byte (#50483) +* [GH-50487](https://github.com/apache/arrow/issues/50487) - [C++] Extra semicolon warning from ARROW_SUPPRESS_DEPRECATION_WARNING macro with -Wpedantic (#50489) +* [GH-50514](https://github.com/apache/arrow/issues/50514) - [R] read_ipc_stream fails to unify nested Enum fields inside a Struct array across record batches (#51153) +* [GH-50532](https://github.com/apache/arrow/issues/50532) - [R][CI] R nightly binary upload broken since github3 to pygithub migration (#50533) +* [GH-50537](https://github.com/apache/arrow/issues/50537) - [R] Fix vector_logic_linter warning by using && in if condition (#50538) +* [GH-50542](https://github.com/apache/arrow/issues/50542) - [C++] Fix ARROW_SIMD_LEVEL=NONE build, compile SSE4.2 kernels (#50547) +* [GH-50579](https://github.com/apache/arrow/issues/50579) - [Python] Fix test_categorical_order_survives_roundtrip pandas 3.X deprecation (#50608) +* [GH-50581](https://github.com/apache/arrow/issues/50581) - [C++][R] Use bundled simdjson for C++ wrapper builds (#50587) +* [GH-50582](https://github.com/apache/arrow/issues/50582) - [CI][C++] Install missing `simdjson-static` on Alpine Linux (#50583) +* [GH-50585](https://github.com/apache/arrow/issues/50585) - [CI][C++] Use bundled simdjson on Alpine Linux (#50586) +* [GH-50591](https://github.com/apache/arrow/issues/50591) - [Python] Fix reference in ConvertToSequenceAndInferSize when an iterator raises on array conversion (#50594) +* [GH-50597](https://github.com/apache/arrow/issues/50597) - [CI] Retry Chrome PyArrow load and fix Snappy Emscripten configure (#50598) +* [GH-50636](https://github.com/apache/arrow/issues/50636) - [C++] Replace std::span/ranges usage to fix macOS CRAN (#50705) +* [GH-50641](https://github.com/apache/arrow/issues/50641) - [C++][Compute] Fix correctness error in decimal round_binary kernel (#50642) +* [GH-50648](https://github.com/apache/arrow/issues/50648) - [Packaging][Linux] Enable OpenTelemetry (#50650) +* [GH-50678](https://github.com/apache/arrow/issues/50678) - [C++][Parquet] Remove unused member `null_slot_usage` in struct `LevelInfo` (#50679) +* [GH-50680](https://github.com/apache/arrow/issues/50680) - [C++][Dev] Extend type IDs in gdb_arrow.py (#50683) +* [GH-50684](https://github.com/apache/arrow/issues/50684) - [Python][FlightRPC] Break the reference cycle between the C++ FlightServerBase and the Python object to avoid leaking server (#50687) +* [GH-50688](https://github.com/apache/arrow/issues/50688) - [CI] Remove brew update to fix macOS arrow-s3fs-test segfaults (#50734) +* [GH-50702](https://github.com/apache/arrow/issues/50702) - [Python] Fix .pyx changes requiring two builds to take effect (#50719) +* [GH-50716](https://github.com/apache/arrow/issues/50716) - [C++][CI] Make simdjson required for Parquet (#50717) +* [GH-50718](https://github.com/apache/arrow/issues/50718) - [C++][CI] Fix valgrind use of uninitialised value on FixedSizeListTestCase (#50721) +* [GH-50730](https://github.com/apache/arrow/issues/50730) - [C++] Do not use throwing api (#50732) +* [GH-50737](https://github.com/apache/arrow/issues/50737) - [C++][Parquet] mark `MakeStatistics` method without `ColumnDescriptor` as deprecated (#50738) +* [GH-50739](https://github.com/apache/arrow/issues/50739) - [C++] Make simdjson required for static linking (#50741) +* [GH-50744](https://github.com/apache/arrow/issues/50744) - [R] Add read_ipc_file and write_ipc_file to _pkgdown.yml reference index (#50745) +* [GH-50750](https://github.com/apache/arrow/issues/50750) - [C++][Parquet] Remove code marked as deprecated except flight in versions 23.0.0 and earlier (#50751) +* [GH-50752](https://github.com/apache/arrow/issues/50752) - [C++][Compute] Fix unused variable warning when ARROW_WITH_RE2 is disabled (#50754) +* [GH-50756](https://github.com/apache/arrow/issues/50756) - [C++][FlightSQL][ODBC] Fix Clang 20 compilation on macOS 26 (#50757) +* [GH-50758](https://github.com/apache/arrow/issues/50758) - [CI][C++] Use LLVM 22 on Debian experimental (#50759) +* [GH-50760](https://github.com/apache/arrow/issues/50760) - [CI][Python] Create venv for test-fedora-42-python-3 (#50761) +* [GH-50774](https://github.com/apache/arrow/issues/50774) - [CI][Python] Match Protobuf symbol visibility in bundled Substrait and ORC (#50792) +* [GH-50778](https://github.com/apache/arrow/issues/50778) - [C++][Parquet] Fix chunked level histogram accumulation (#50780) +* [GH-50811](https://github.com/apache/arrow/issues/50811) - [Release] Use maint-Major.Minor.x for patch releases as the maintenance branch on required release scripts (#50813) +* [GH-50819](https://github.com/apache/arrow/issues/50819) - [Release] Increase YUM verification timeout (#50835) +* [GH-50849](https://github.com/apache/arrow/issues/50849) - [Python] Return correct ParquetLogicalType.type for geometry/geography (#50850) +* [GH-50859](https://github.com/apache/arrow/issues/50859) - [C++][Parquet] Move JsonWriter to simdjson utilities (#50990) +* [GH-50862](https://github.com/apache/arrow/issues/50862) - [C++][Gandiva] Fix Gandiva tests on riscv64 with an LLVM JIT relocation error (#50799) +* [GH-50868](https://github.com/apache/arrow/issues/50868) - [C++][CI] Suppress deprecated Abseil API warnings for GCS on macOS (#50887) +* [GH-50869](https://github.com/apache/arrow/issues/50869) - [C++][Compute] Tighten coalesce exact dispatch for decimal varargs (#50870) +* [GH-50971](https://github.com/apache/arrow/issues/50971) - [C++][Parquet] Fix usage of disparate length types for metadata reading (#50972) +* [GH-50985](https://github.com/apache/arrow/issues/50985) - [CI][Dev][Python] Update cython-lint and pin Cython to 3.2.9 (#50986) +* [GH-50987](https://github.com/apache/arrow/issues/50987) - [C++] Fix builds with the macOS 11.3 SDK (#50997) +* [GH-50988](https://github.com/apache/arrow/issues/50988) - [C++][Python] Substrait: add mappings for starts_with, ends_with and match_substring (#50989) +* [GH-50991](https://github.com/apache/arrow/issues/50991) - [CI] Enable sccache just like ccache on Docker-based builds (#50992) +* [GH-50999](https://github.com/apache/arrow/issues/50999) - [CI][Python] Fix NuGet based vcpkg cache failures on musllinux (#51092) +* [GH-51007](https://github.com/apache/arrow/issues/51007) - [C++] Make uriparser an external dependency (#51244) +* [GH-51011](https://github.com/apache/arrow/issues/51011) - [C++][CI] Poll for GCS testbench readiness instead of one 10s attempt (#51012) +* [GH-51041](https://github.com/apache/arrow/issues/51041) - [Python] Reject non-Buffer FunctionOptions.deserialize input (#51129) +* [GH-51044](https://github.com/apache/arrow/issues/51044) - [Python] Reject read-only readinto destinations (#51126) +* [GH-51095](https://github.com/apache/arrow/issues/51095) - [CI][C++] Fix core file detection in run-test.sh (#51121) +* [GH-51098](https://github.com/apache/arrow/issues/51098) - [Release] Update CHANGELOG.md as a post release task (#51117) +* [GH-51101](https://github.com/apache/arrow/issues/51101) - [C++][Emscripten] Increase test stack size (#51103) +* [GH-51135](https://github.com/apache/arrow/issues/51135) - [C++][Parquet] Avoid misaligned stores when reading BYTE_ARRAY decimals (#51136) +* [GH-51138](https://github.com/apache/arrow/issues/51138) - [Python] Fix ccache efficiency (#51139) +* [GH-51140](https://github.com/apache/arrow/issues/51140) - [C++] Require simdjson 4.3.0 for fractured_json APIs (#51142) +* [GH-51144](https://github.com/apache/arrow/issues/51144) - [C++] Incorrect Logic for AdaptiveUIntBuilder::AppendValues (#51146) +* [GH-51152](https://github.com/apache/arrow/issues/51152) - [R] test-r-linux-as-cran nightly fails with NOTE about non-standard top-level file jarl.toml (#51154) +* [GH-51156](https://github.com/apache/arrow/issues/51156) - [C++] Raise CapacityError instead of truncating binary values over 2 GiB (#51158) +* [GH-51194](https://github.com/apache/arrow/issues/51194) - [C++] Fix cumulative_max/min default start for floating-point types (#51203) +* [GH-51210](https://github.com/apache/arrow/issues/51210) - [C++] Initialize `output_` on empty `select_k` inputs (#51212) +* [GH-51217](https://github.com/apache/arrow/issues/51217) - [CI][Packaging] Don't use diffoscope (#51221) +* [GH-51229](https://github.com/apache/arrow/issues/51229) - [Python] Raise instead of crashing on unopened resize (#51246) +* [GH-51254](https://github.com/apache/arrow/issues/51254) - [C++][CI] Move ubuntu-cpp-bundled-offline to C++ Extra (#51255) +* [GH-51285](https://github.com/apache/arrow/issues/51285) - [CI] Fix AMD64 Conda Integration Test failure (#51287) +* [GH-51327](https://github.com/apache/arrow/issues/51327) - [CI][Dev] Change download minIO URLs for GitHub releases URL (#51326) +* [GH-51335](https://github.com/apache/arrow/issues/51335) - [CI] Update Matlab actions and fix Matlab Windows failure (#51378) +* [GH-51343](https://github.com/apache/arrow/issues/51343) - [C++][CI][Packaging] Disable Precompile Headers on simdjson to avoid not reproducible artifacts (#51355) +* [GH-51345](https://github.com/apache/arrow/issues/51345) - [CI] Fix pipx install for `packaging` on Debian (#51346) +* [GH-51361](https://github.com/apache/arrow/issues/51361) - [C++][Parquet] Derive records_to_read in FileReaderImpl::ReadColumn from RowGroup (#51362) +* [GH-51370](https://github.com/apache/arrow/issues/51370) - [C++][Parquet] Fix tracing column attributes (#51401) +* [GH-51402](https://github.com/apache/arrow/issues/51402) - [CI][Release] Bump macos runner versions to supported version from Homebrew (#51403) +* [GH-51446](https://github.com/apache/arrow/issues/51446) - [CI][Python] Fix test collection errors when Parquet encryption is unavailable (#51457) +* [GH-51456](https://github.com/apache/arrow/issues/51456) - [C++] Make ABI independent of ARROW_EXTRA_ERROR_CONTEXT (#51458) +* [GH-51467](https://github.com/apache/arrow/issues/51467) - [CI][C++] Remove unused ranges include which breaks R build (#51469) +* [GH-51487](https://github.com/apache/arrow/issues/51487) - [CI] Fix Emscripten wheel load failure by disabling llvm-strip on install (#51488) +* [GH-51492](https://github.com/apache/arrow/issues/51492) - [C++] Do not left shift a bpacking lane by its full width (#51493) +* [GH-51494](https://github.com/apache/arrow/issues/51494) - [CI][C++] Detect Visual Studio installed path instead of hardcoding it (#51610) +* [GH-51495](https://github.com/apache/arrow/issues/51495) - [C++] Fix race in MergedGenerator that could drop an error and end the stream early (#51498) +* [GH-51496](https://github.com/apache/arrow/issues/51496) - [C++] Fix 32-bit block offset overflows in SwissTable (#51497) +* [GH-51604](https://github.com/apache/arrow/issues/51604) - [C++][Parquet] Only link opentelemetry-cpp::sdk on arrow_reader_writer_tracing_test.cc to avoid double freeing (#51606) +* [GH-51632](https://github.com/apache/arrow/issues/51632) - [C++][CI] Link OpenTelemetry libs for parquet-arrow-reader-writer-tracing-test depending on System vs Bundled OpenTelemetry (#51633) +* [GH-51634](https://github.com/apache/arrow/issues/51634) - [C++][CI] Apply a google-cloud-cpp patch for OpenSSL 4.x compatibility (#51635) +* [GH-51638](https://github.com/apache/arrow/issues/51638) - [CI][C++][Parquet] Update expected JSON output in parquet-reader-test for simdjson 5.0 (#51661) +* [GH-51643](https://github.com/apache/arrow/issues/51643) - [CI] Run extra jobs on pushes to main again (#51666) +* [GH-51651](https://github.com/apache/arrow/issues/51651) - [C++][CI] Fix static linking for system Abseil and bundled GCS (#51653) +* [GH-51675](https://github.com/apache/arrow/issues/51675) - [Python][Packaging] Apply upstream fix to UriParser from vcpkg and force Windows image rebuild to fix builds on wheels (#51676) +* [GH-51682](https://github.com/apache/arrow/issues/51682) - [CI][C++] Add liburiparser-dev to system dependency image (#51683) +* [GH-51687](https://github.com/apache/arrow/issues/51687) - [C++] Skipt tests that require threads if ARROW_ENABLE_THREADING=OFF (#51688) + + +## New Features and Improvements + +* [GH-12594](https://github.com/apache/arrow/issues/12594) - [Python][GPU] Remove inherently crashy CUDA test (#51206) +* [GH-14734](https://github.com/apache/arrow/issues/14734) - [R] Deprecated filter + across usage (#51235) +* [GH-30800](https://github.com/apache/arrow/issues/30800) - [Python][Docs] Document partition fields with explicit dataset schemas (#50352) +* [GH-32123](https://github.com/apache/arrow/issues/32123) - [R] Expose azure blob filesystem (#49553) +* [GH-33708](https://github.com/apache/arrow/issues/33708) - [R] read_csv_arrow()'s timestamp_parsers parameter is a bit light on documentation and doesn't appear to do anything (#51166) +* [GH-34577](https://github.com/apache/arrow/issues/34577) - [Python] Expose eol and null_string csv WriteOptions (#46976) +* [GH-36010](https://github.com/apache/arrow/issues/36010) - [GLib][Ruby][Parquet] Add buffered reader properties (#51277) +* [GH-37853](https://github.com/apache/arrow/issues/37853) - [Python] Remove test and fixture involving fastparquet (#50416) +* [GH-38771](https://github.com/apache/arrow/issues/38771) - [R][Documentation] Document add_filename on open_dataset help page (#51392) +* [GH-38868](https://github.com/apache/arrow/issues/38868) - [Python] Add dlpack producer to FixedShapeTensorArray/Scalar (#51159) +* [GH-38868](https://github.com/apache/arrow/issues/38868) - [C++][Python] Add Array::ToTensor and fixed size list support (#50929) +* [GH-39295](https://github.com/apache/arrow/issues/39295) - [C++][Python] ConsumingDLPack on Arrays and Tensor (#51122) +* [GH-39660](https://github.com/apache/arrow/issues/39660) - [R] Document partition value types (#50857) +* [GH-40282](https://github.com/apache/arrow/issues/40282) - [Python] Use C++ type traits for is_nested function (#41709) +* [GH-41670](https://github.com/apache/arrow/issues/41670) - [C++][Python] Move to DLPack 1.3 (#50827) +* [GH-45747](https://github.com/apache/arrow/issues/45747) - [C++] Remove deprecated ObjectType and FileStatistics, refactor hdfs code (#45998) +* [GH-46853](https://github.com/apache/arrow/issues/46853) - [Python][Docs] Document preserving leading zeros in CSV (#50477) +* [GH-46901](https://github.com/apache/arrow/issues/46901) - [C++][Compute] Add remainder and modulo kernels (#48914) +* [GH-47390](https://github.com/apache/arrow/issues/47390) - [C++][Acero] Allow for any type of scalar in Pivot longer features (#47391) +* [GH-47402](https://github.com/apache/arrow/issues/47402) - [CI][Dev] Fix shellcheck errors in the ci/scripts/python_test_emscripten.sh (#47403) +* [GH-47472](https://github.com/apache/arrow/issues/47472) - [Doc] List third-party implementations (#51390) +* [GH-47499](https://github.com/apache/arrow/issues/47499) - [Python] Explicitly add `use_content_defined_chunking` to `ParquetWriter` and `write_table` (#47498) +* [GH-47583](https://github.com/apache/arrow/issues/47583) - [CI][Python] Enable PyArrow RelWithDebInfo build with assertions on Windows CI job (#50406) +* [GH-47686](https://github.com/apache/arrow/issues/47686) - [Docs][Python] Split the Python Parquet docs into separate items (#50419) +* [GH-47877](https://github.com/apache/arrow/issues/47877) - [Packaging][C++][FlightRPC][ODBC] Add arrow-flight-sql-odbc (#50288) +* [GH-48172](https://github.com/apache/arrow/issues/48172) - [Python] Add cp315 to build (#48191) +* [GH-48222](https://github.com/apache/arrow/issues/48222) - [CI][Dev] Fix shellcheck errors in ci/scripts/cpp_build.sh (#48223) +* [GH-48473](https://github.com/apache/arrow/issues/48473) - [CI][Python] Require numpy 2.0 (#50769) +* [GH-48740](https://github.com/apache/arrow/issues/48740) - [C++] Add missing CTypeTraits for decimal types (#50153) +* [GH-48743](https://github.com/apache/arrow/issues/48743) - [C++] Reenable timezone tests on Windows GCC (#51211) +* [GH-48808](https://github.com/apache/arrow/issues/48808) - [Python] Drop support for Pandas < 2.0.3 (#50444) +* [GH-49046](https://github.com/apache/arrow/issues/49046) - [Dev][Python] Remove unused scripts under python/scripts (#50640) +* [GH-49231](https://github.com/apache/arrow/issues/49231) - [C++] Deprecate Feather reader and writer (#50321) +* [GH-49237](https://github.com/apache/arrow/issues/49237) - [R] Deprecate Feather reader and writer (#49276) +* [GH-49255](https://github.com/apache/arrow/issues/49255) - [Python] Fix pandas Categorical DeprecationWarnings in tests (#50543) +* [GH-49305](https://github.com/apache/arrow/issues/49305) - [Python] Expose RecordBatchFileReader.count_rows (#50646) +* [GH-49524](https://github.com/apache/arrow/issues/49524) - [CI][Integration] Bump HDFS versions tested to use latest v2 and newest v3 (#50814) +* [GH-49538](https://github.com/apache/arrow/issues/49538) - [C++][FlightRPC][ODBC] Use static linkage in Windows FlightSQL ODBC driver (#49585) +* [GH-49677](https://github.com/apache/arrow/issues/49677) - [Python][C++][Compute] Add search sorted compute kernel (#49679) +* [GH-49970](https://github.com/apache/arrow/issues/49970) - [GLib] Enable tests for custom extension data type (#49971) +* [GH-49977](https://github.com/apache/arrow/issues/49977) - [C++][Gandiva] Add regexp_extract optional third parameter function version (#49978) +* [GH-50087](https://github.com/apache/arrow/issues/50087) - [Docs][C++] Fix sentence structure in memory.rst regarding `MemoryManager` (#50324) +* [GH-50091](https://github.com/apache/arrow/issues/50091) - [Packaging][Python] Add support for Python 3.15 +* [GH-50136](https://github.com/apache/arrow/issues/50136) - [C++][Gandiva] Enhance CHR to work with unicode (#50137) +* [GH-50194](https://github.com/apache/arrow/issues/50194) - [C++] Move S3 and AWS-SDK to its own libarrow_s3.so (#50195) +* [GH-50223](https://github.com/apache/arrow/issues/50223) - [C++][Compute] Support string_view/binary_view keys in the hash-aggregate Grouper (#50224) +* [GH-50247](https://github.com/apache/arrow/issues/50247) - [C++] Reuse abstraction for null partitions in sorting functions (#50248) +* [GH-50250](https://github.com/apache/arrow/issues/50250) - [C++] Remove `call_traits::argument_type` in favor of `` (#51328) +* [GH-50251](https://github.com/apache/arrow/issues/50251) - [C++] Add GetSpan to ArrayData (#50366) +* [GH-50280](https://github.com/apache/arrow/issues/50280) - [C++] Implement VisitTwoBitRuns and VisitTwoSetBitRuns methods (#50281) +* [GH-50298](https://github.com/apache/arrow/issues/50298) - [CI][R] Update ubuntu-clang CI job to use clang 22 to match CRAN (#50299) +* [GH-50313](https://github.com/apache/arrow/issues/50313) - [C++][Docs] Add guidance about memory bombs (#50408) +* [GH-50333](https://github.com/apache/arrow/issues/50333) - [C++][Parquet] Add dense decode path for FIXED_LEN_BYTE_ARRAY (#50335) +* [GH-50338](https://github.com/apache/arrow/issues/50338) - [C++] Add ComputeLogicalNullCount to Datum (#50347) +* [GH-50345](https://github.com/apache/arrow/issues/50345) - [CI] Remove redundant checkout step from check_labels.yml (#50346) +* [GH-50355](https://github.com/apache/arrow/issues/50355) - [C++][Gandiva] fix out-of-bounds read in utf8_length_ignore_invalid (#50356) +* [GH-50379](https://github.com/apache/arrow/issues/50379) - [Dev] Convert invalid PRs to draft automatically (#50467) +* [GH-50380](https://github.com/apache/arrow/issues/50380) - [C++][Gandiva] fix out-of-bounds read in byte_substr past end (#50381) +* [GH-50385](https://github.com/apache/arrow/issues/50385) - [Ruby] Add bitmap builder for red-arrow-format (#50386) +* [GH-50390](https://github.com/apache/arrow/issues/50390) - [Ruby] Add int/float array builder for red-arrow-format (#50391) +* [GH-50394](https://github.com/apache/arrow/issues/50394) - [Docs][C++] Reduce Sphinx warnings when building HTML (#50396) +* [GH-50395](https://github.com/apache/arrow/issues/50395) - [C++] Support duration inputs in temporal rounding (#50675) +* [GH-50410](https://github.com/apache/arrow/issues/50410) - [Python][Packaging] Drop Python 3.10 support (#50411) +* [GH-50412](https://github.com/apache/arrow/issues/50412) - [CI][GLib][Ruby] Remove some unnecessary Ubuntu 20.04 cases (#50413) +* [GH-50417](https://github.com/apache/arrow/issues/50417) - [CI][Dev] Fix shellcheck error in ci/scripts/install_bison.sh (#50418) +* [GH-50421](https://github.com/apache/arrow/issues/50421) - [C++][Parquet] Add LevelDecoder Skip and Count (#50422) +* [GH-50431](https://github.com/apache/arrow/issues/50431) - [Python] Make FlightError a non-cdef class for abi3 wheel precursor (#50427) +* [GH-50432](https://github.com/apache/arrow/issues/50432) - [Ruby] Add `ArrowFormat::{Array,Bitmap}#==` (#50433) +* [GH-50434](https://github.com/apache/arrow/issues/50434) - [Ruby] Add `ArrowFormat::Date{32,64}.new(values)` (#50442) +* [GH-50435](https://github.com/apache/arrow/issues/50435) - [Ruby] Add `ArrowFormat::Time{32,64}.new(unit, values)` (#50443) +* [GH-50436](https://github.com/apache/arrow/issues/50436) - [Ruby] Add `ArrowFormat::TimestampArray.new(unit, values)` (#50445) +* [GH-50437](https://github.com/apache/arrow/issues/50437) - [Ruby] Add `ArrowFormat::*IntervalArray.new(values)` (#50459) +* [GH-50438](https://github.com/apache/arrow/issues/50438) - Add ArrowFormat::DurationArray.new(unit, values) (#50447) +* [GH-50439](https://github.com/apache/arrow/issues/50439) - Add ArrowFormat::{,Large}{Binary,UTF8}Array.new(values) (#50452) +* [GH-50440](https://github.com/apache/arrow/issues/50440) - [C++][Gandiva] Fix out-of-bounds read in set_error_for_date (#50441) +* [GH-50455](https://github.com/apache/arrow/issues/50455) - [Ruby] Add `ArrowFormat::TimeType#==` (#50461) +* [GH-50456](https://github.com/apache/arrow/issues/50456) - [Ruby] Add `ArrowFormat::DurationType#==` (#50460) +* [GH-50457](https://github.com/apache/arrow/issues/50457) - [C++][Gandiva] Add LN alias for LOG (#50458) +* [GH-50462](https://github.com/apache/arrow/issues/50462) - [C++][Gandiva] fix out-of-bounds read in translate_utf8_utf8_utf8 (#50463) +* [GH-50464](https://github.com/apache/arrow/issues/50464) - [C++][Python] Simplify arrow_to_pandas DateOffset handling for nanoseconds/milliseconds (#50465) +* [GH-50492](https://github.com/apache/arrow/issues/50492) - [R] Add release process skill (#50499) +* [GH-50493](https://github.com/apache/arrow/issues/50493) - [Python] Use scikit-build-core force-include and remove custom build-backend to copy license files (#50494) +* [GH-50495](https://github.com/apache/arrow/issues/50495) - [R] 25.0.0 Release followups (#50786) +* [GH-50496](https://github.com/apache/arrow/issues/50496) - [R] Polish NEWS.md for 25.0.0 (#50497) +* [GH-50508](https://github.com/apache/arrow/issues/50508) - [C++] Support scalar values in AppendScalars (#50584) +* [GH-50510](https://github.com/apache/arrow/issues/50510) - [CI][Dev] Fix shellcheck error in the ci/scripts/ccache_fix_perms.sh (#50511) +* [GH-50512](https://github.com/apache/arrow/issues/50512) - [C++][Compute] Support float16 in hash kernels (dictionary_encode, unique, value_counts) (#50513) +* [GH-50518](https://github.com/apache/arrow/issues/50518) - [Packaging][C++][FlightRPC][ODBC][RPM] Add suppport for auto driver registration (#50521) +* [GH-50519](https://github.com/apache/arrow/issues/50519) - [C++][FlightRPC][ODBC] Add missing `ARROW_FLIGHT_SQL_ODBC_INSTALLER` option entry (#50520) +* [GH-50524](https://github.com/apache/arrow/issues/50524) - [C++] Honor array offset in pairwise_diff (#50858) +* [GH-50528](https://github.com/apache/arrow/issues/50528) - [Ruby] Add `ArrowFormat::ArrayBuilder` (#50529) +* [GH-50531](https://github.com/apache/arrow/issues/50531) - [Python][Packaging] Set macOS deployment target before building wheel platform tag (#50377) +* [GH-50535](https://github.com/apache/arrow/issues/50535) - [CI][Dev] Fix shellcheck errors in the ci/scripts/python_test.sh (#50536) +* [GH-50551](https://github.com/apache/arrow/issues/50551) - [CI][Dev] Fix shellcheck errors in the ci/scripts/python_wheel_macos_build.sh (#50552) +* [GH-50553](https://github.com/apache/arrow/issues/50553) - [Dev] Clarify invalid PR title guidance comment (#50554) +* [GH-50560](https://github.com/apache/arrow/issues/50560) - [C++][FlightRPC][ODBC] Fix SQLDescribeCol column_size/decimal_digits width (#50562) +* [GH-50565](https://github.com/apache/arrow/issues/50565) - [CI][C++] Add a JNI Windows CI job (#50606) +* [GH-50566](https://github.com/apache/arrow/issues/50566) - [C++] Introduce simdjson and migrate ObjectParser (#50469) +* [GH-50567](https://github.com/apache/arrow/issues/50567) - [C++] Introduce JsonWriter and migrate integration JSON writer (#50568) +* [GH-50569](https://github.com/apache/arrow/issues/50569) - [Ruby] Add `ArrowFormat::RecordBatch.new(values)` (#50570) +* [GH-50571](https://github.com/apache/arrow/issues/50571) - [Ruby] Add schema-aware ArrowFormat::RecordBatch construction (#51214) +* [GH-50572](https://github.com/apache/arrow/issues/50572) - [Ruby] Add `ArrowFormat::RecordBatch#records` (#50590) +* [GH-50573](https://github.com/apache/arrow/issues/50573) - [CI][C++] Remove brew workaround for aws-sdk-cpp (#50557) +* [GH-50575](https://github.com/apache/arrow/issues/50575) - [CI][Dev] Fix shellcheck errors in the ci/scripts/python_wheel_xlinux_build.sh (#50577) +* [GH-50588](https://github.com/apache/arrow/issues/50588) - [Ruby] ` (#50589) +* [GH-50601](https://github.com/apache/arrow/issues/50601) - [C++] Refactor TranslateTo for clearer semantics (#50602) +* [GH-50609](https://github.com/apache/arrow/issues/50609) - [CI][Dev] Fix shellcheck errors in the ci/scripts/r_deps.sh (#50610) +* [GH-50615](https://github.com/apache/arrow/issues/50615) - [C++] Reduce code generation for string kernels (#50616) +* [GH-50622](https://github.com/apache/arrow/issues/50622) - [Docs][Format] Align Variant `typed_value` primitive type mappings with the Parquet shredding spec (#50810) +* [GH-50623](https://github.com/apache/arrow/issues/50623) - [C++][IPC] Fix extension-wrapped union IPC roundtrip (#50927) +* [GH-50624](https://github.com/apache/arrow/issues/50624) - [C++][Compute] Tighten case_when exact dispatch for parameterized types (#50625) +* [GH-50627](https://github.com/apache/arrow/issues/50627) - [C++] Migrate from_string.cc to simdjson (#50653) +* [GH-50631](https://github.com/apache/arrow/issues/50631) - [C++] Scope macOS SDK 11.3 simdjson workaround to bundled simdjson (#50633) +* [GH-50654](https://github.com/apache/arrow/issues/50654) - [C++] Support simdjson without exceptions (#50672) +* [GH-50660](https://github.com/apache/arrow/issues/50660) - [C++][Dev] Add Decimal32 and Decimal64 GDB pretty-printers (#50723) +* [GH-50670](https://github.com/apache/arrow/issues/50670) - [Release][Dev] Fix only shellcheck SC2086 errors in the dev directory (#50671) +* [GH-50674](https://github.com/apache/arrow/issues/50674) - [Release] Add support for recovering Yum repositories (#50681) +* [GH-50690](https://github.com/apache/arrow/issues/50690) - [C++] Migrate ObjectWriter users to JsonWriter (#50691) +* [GH-50697](https://github.com/apache/arrow/issues/50697) - [C++][FlightRPC] ODBC installer support fixes (#50748) +* [GH-50706](https://github.com/apache/arrow/issues/50706) - [C++] Migrate extension type serialization to JsonWriter (#50708) +* [GH-50707](https://github.com/apache/arrow/issues/50707) - [C++][Gandiva] fix out-of-bounds read in mask utf8proc length (#50709) +* [GH-50713](https://github.com/apache/arrow/issues/50713) - [C++] Replace `return_type` and `enable_if_return` with helpers (#50714) +* [GH-50724](https://github.com/apache/arrow/issues/50724) - [C++] Add `JsonWriter::WriteValue` for simdjson values (#50725) +* [GH-50728](https://github.com/apache/arrow/issues/50728) - [C++] Remove `call_traits::argument_count` in favor of `type_traits` (#50729) +* [GH-50766](https://github.com/apache/arrow/issues/50766) - [CI][Dev] Fix shellcheck errors in the ci/scripts/r_docker_configure.sh (#50767) +* [GH-50771](https://github.com/apache/arrow/issues/50771) - [CI][Dev] Fix shellcheck errors in the ci/scripts/r_install_system_dependencies.sh (#50772) +* [GH-50773](https://github.com/apache/arrow/issues/50773) - [CI][Dev] Fix shellcheck errors in the ci/scripts/r_sanitize.sh (#50775) +* [GH-50777](https://github.com/apache/arrow/issues/50777) - [CI][Dev] Fix shellcheck errors in the ci/scripts/r_test.sh (#50795) +* [GH-50779](https://github.com/apache/arrow/issues/50779) - [C++][Parquet] Replace remaining RapidJSON usage with simdjson (#50781) +* [GH-50784](https://github.com/apache/arrow/issues/50784) - [C++] Align write in `TransferBitmap` (#50785) +* [GH-50796](https://github.com/apache/arrow/issues/50796) - [CI][Dev] Fix shellcheck errors in the ci/scripts/r_valgrind.sh (#50798) +* [GH-50801](https://github.com/apache/arrow/issues/50801) - [Python] Expose the `record_batch_reader_source` Acero node (RecordBatchReaderSourceNodeOptions) (#50802) +* [GH-50803](https://github.com/apache/arrow/issues/50803) - [CI][Dev] Fix shellcheck errors in the ci/scripts/r_windows_build.sh (#50818) +* [GH-50820](https://github.com/apache/arrow/issues/50820) - [CI][Dev] Bump ShellCheck to v0.11.0 and simplify pre-commit file patterns for the ci directory (#50821) +* [GH-50824](https://github.com/apache/arrow/issues/50824) - [R] Fix shellcheck errors in the r/tools/download_dependencies_R.sh (#50825) +* [GH-50830](https://github.com/apache/arrow/issues/50830) - [C++][Parquet] Use JsonWriter for LogicalType::ToJSON() (#50877) +* [GH-50832](https://github.com/apache/arrow/issues/50832) - [Ruby] Add `ArrowFormat::FixedSizeBinaryArray.new(byte_width, values)` (#50854) +* [GH-50840](https://github.com/apache/arrow/issues/50840) - [C++] Fix dead overflow guard in Take on binary-like arrays (#50841) +* [GH-50847](https://github.com/apache/arrow/issues/50847) - [Python] Add Type_RUN_END_ENCODED to _NESTED_TYPES set (#50848) +* [GH-50855](https://github.com/apache/arrow/issues/50855) - [R] Fix shellcheck errors in the r/inst/build_arrow_static.sh (#50856) +* [GH-50860](https://github.com/apache/arrow/issues/50860) - [Dev][R][Swift][GLib] Simplify pre-commit file patterns for the r, swift, and c_glib directories (#50861) +* [GH-50876](https://github.com/apache/arrow/issues/50876) - [C++][FS][Azure] Wait for Azurite to start (#50878) +* [GH-50883](https://github.com/apache/arrow/issues/50883) - [C++] Remove unreferenced cpp/build-support/trim-boost.sh (#50885) +* [GH-50897](https://github.com/apache/arrow/issues/50897) - [Python] Fix typo LZ0 -> LZO in ORC writer docstring (#50898) +* [GH-50899](https://github.com/apache/arrow/issues/50899) - [CI][C++] Use OIDC instead of ACCESS_KEY and SECRET_KEY for sccache's s3 creds (#51028) +* [GH-50901](https://github.com/apache/arrow/issues/50901) - [C++] Replace RapidJSON with simdjson in tensor extension types (#50874) +* [GH-50904](https://github.com/apache/arrow/issues/50904) - [C++] Replace RapidJSON with simdjson in OpaqueType (#50905) +* [GH-50910](https://github.com/apache/arrow/issues/50910) - [Parquet] Tolerate unrecognized logical/physical type combinations when reading (#50909) +* [GH-50911](https://github.com/apache/arrow/issues/50911) - [C++][FlightSQL][ODBC] Replace RapidJSON with JsonWriter (#50912) +* [GH-50913](https://github.com/apache/arrow/issues/50913) - [C++][Dataset] Replace RapidJSON with JsonWriter (#50914) +* [GH-50917](https://github.com/apache/arrow/issues/50917) - [C++] Fix shellcheck errors in cpp/build-support/*-flatbuffers.sh (#50918) +* [GH-50925](https://github.com/apache/arrow/issues/50925) - [C++] Allow CSV reader to pad rows with missing trailing fields (#50926) +* [GH-50934](https://github.com/apache/arrow/issues/50934) - [C++][Dev] Fix shellcheck errors in cpp/build-support/run-test.sh (#50935) +* [GH-50936](https://github.com/apache/arrow/issues/50936) - [C++][Integration] Replace RapidJSON with simdjson (#50937) +* [GH-50943](https://github.com/apache/arrow/issues/50943) - [Python] Add Pandas nightly's for cp315 & cp315t tests (#50946) +* [GH-50944](https://github.com/apache/arrow/issues/50944) - [C++] Replace RapidJSON with simdjson in JSON chunker (#50945) +* [GH-50963](https://github.com/apache/arrow/issues/50963) - [C++] Update bundled bzip2 WrapDB package (#50924) +* [GH-50967](https://github.com/apache/arrow/issues/50967) - [C++] Allow CSV reader to ignore extra columns in rows with more columns (#51118) +* [GH-50978](https://github.com/apache/arrow/issues/50978) - [C++] Enable MSVC's new preprocessor to handle C++20 grammar (#50979) +* [GH-50993](https://github.com/apache/arrow/issues/50993) - [CI][Integration] Add extension-wrapped union to integration data (#51027) +* [GH-51003](https://github.com/apache/arrow/issues/51003) - [CI][Dev][Python] Bump cython-lint to 0.21.1 and remove Cython pin (#51202) +* [GH-51005](https://github.com/apache/arrow/issues/51005) - [C++] Fix over-read in UriFromAbsolutePath posix branch (#51006) +* [GH-51013](https://github.com/apache/arrow/issues/51013) - [C++] Replace RapidJSON in JSON test utilities (#51014) +* [GH-51029](https://github.com/apache/arrow/issues/51029) - [C++] Fix MapArray validation with unknown null count (#51093) +* [GH-51039](https://github.com/apache/arrow/issues/51039) - [CI][GLib][Ruby] Re-enable Arrow Flight tests for GLib and Ruby on macOS (#51040) +* [GH-51090](https://github.com/apache/arrow/issues/51090) - [CI][C++] Fix shellcheck errors in cpp/build-support/fuzzing/generate_corpuses.sh (#51091) +* [GH-51096](https://github.com/apache/arrow/issues/51096) - [CI][C++] Remove useless test-file-cleanup and retry logic from run-test.sh (#51110) +* [GH-51097](https://github.com/apache/arrow/issues/51097) - Fix Parquet null counts for fixed-width leaves (#51357) +* [GH-51100](https://github.com/apache/arrow/issues/51100) - [C++][Format] Deprecate IPC tensor/sparse tensor messages (#51102) +* [GH-51111](https://github.com/apache/arrow/issues/51111) - [Dev] Group apache/infrastructure-actions/stash/{save,restore} Dependabot updates (#51112) +* [GH-51116](https://github.com/apache/arrow/issues/51116) - [C++] Update bundled Boost (#51132) +* [GH-51127](https://github.com/apache/arrow/issues/51127) - [C++] Fix is_null nan_is_null for dictionary-encoded float arrays (#51000) +* [GH-51145](https://github.com/apache/arrow/issues/51145) - [Python] Accept Arrow integer scalars in slice methods (#51160) +* [GH-51148](https://github.com/apache/arrow/issues/51148) - [Docs] Repoint three dead links (#51149) +* [GH-51199](https://github.com/apache/arrow/issues/51199) - [CI] Limit concurrent PRs from new contributors (#51200) +* [GH-51207](https://github.com/apache/arrow/issues/51207) - [CI][GPU] Improve compilation caching on CUDA CI jobs (#51213) +* [GH-51208](https://github.com/apache/arrow/issues/51208) - [CI][Python] Update builds to CPython 3.15.0rc2 (cp315) (#51209) +* [GH-51220](https://github.com/apache/arrow/issues/51220) - [Python][Parquet] Accept write_table options in dataset writer (#51461) +* [GH-51223](https://github.com/apache/arrow/issues/51223) - [C++] Fix copying sliced boolean arrays (#51240) +* [GH-51224](https://github.com/apache/arrow/issues/51224) - [C++] Keep the correct nulls when winsorizing a sliced array (#51251) +* [GH-51225](https://github.com/apache/arrow/issues/51225) - [C++] utf8_normalize: compose for NFC and NFKC (#51237) +* [GH-51232](https://github.com/apache/arrow/issues/51232) - [CI] Remove the "take" comment bot for self-assigning issues (#51233) +* [GH-51238](https://github.com/apache/arrow/issues/51238) - [C++][Python][Parquet] Limit schema nesting depth when reading (#51239) +* [GH-51245](https://github.com/apache/arrow/issues/51245) - [C++][Parquet][FOLLOWUP] Add support for LLVM 23.1 (#51468) +* [GH-51245](https://github.com/apache/arrow/issues/51245) - [C++][Compute][Gandiva] Add support for LLVM 23.1 (#51266) +* [GH-51252](https://github.com/apache/arrow/issues/51252) - [C++] Reduce boilerplate in OpaqueType JSON deserialization (#51253) +* [GH-51262](https://github.com/apache/arrow/issues/51262) - [Ruby] Add ListArray values constructor (#51263) +* [GH-51264](https://github.com/apache/arrow/issues/51264) - [C++][Parquet] Make MemoryPool settable on ReaderProperties (#51269) +* [GH-51271](https://github.com/apache/arrow/issues/51271) - [Ruby] Fix Date values in ArrowFormat::Date32Array (#51272) +* [GH-51273](https://github.com/apache/arrow/issues/51273) - [Ruby] Add FixedSizeListArray values constructor (#51274) +* [GH-51278](https://github.com/apache/arrow/issues/51278) - [C++] Fix build error in acero tests with gcc 16 (#51279) +* [GH-51283](https://github.com/apache/arrow/issues/51283) - [CI][Dev] Fix shellcheck errors in cpp/thirdparty/download_dependencies.sh (#51284) +* [GH-51297](https://github.com/apache/arrow/issues/51297) - [Dev] Fix SC2035 shellcheck errors in the cpp/src/arrow/vendored directory (#51298) +* [GH-51300](https://github.com/apache/arrow/issues/51300) - Fix deprecation warnings for `null_placement` (#51307) +* [GH-51302](https://github.com/apache/arrow/issues/51302) - [Python] Avoid deprecated .values call in pandas->pyarrow conversion (#51484) +* [GH-51303](https://github.com/apache/arrow/issues/51303) - [C++][Parquet] Update parquet.thrift to sync with 2.14.0 (#51304) +* [GH-51329](https://github.com/apache/arrow/issues/51329) - [C++][CI] Test static linking with S3 and fix Azure/GCS dependencies on static builds (#51280) +* [GH-51331](https://github.com/apache/arrow/issues/51331) - [CI] Exempt collaborators from concurrent PR limit (#51332) +* [GH-51349](https://github.com/apache/arrow/issues/51349) - [C++][Parquet] Add flag to control pass-through of KMS URLs from key material (#51350) +* [GH-51470](https://github.com/apache/arrow/issues/51470) - [C++] Delimit JSON documents without parsing them (#51472) +* [GH-51471](https://github.com/apache/arrow/issues/51471) - [C++] Disable simdjson threading in chunker (#51473) +* [GH-51485](https://github.com/apache/arrow/issues/51485) - [C++] Fix SC2086 errors in cpp/examples/tutorial_examples directory (#51486) +* [GH-51636](https://github.com/apache/arrow/issues/51636) - [CI][Python] Switch back to Pandas release for Python 3.15 (#51637) +* [GH-51652](https://github.com/apache/arrow/issues/51652) - [Python] Add NumPy StringDType to Arrow conversion (#51157) +* [GH-51655](https://github.com/apache/arrow/issues/51655) - [R] Advance long-standing deprecations in map_batches() and pull() (#51658) +* [GH-51656](https://github.com/apache/arrow/issues/51656) - [R] Polish NEWS.md and README for 26.0.0 (#51657) + + + # Apache Arrow 25.0.1 (2026-08-07 00:00:00+00:00) ## Bug Fixes diff --git a/c_glib/meson.build b/c_glib/meson.build index ba764359d95e..1d92f6c3c00a 100644 --- a/c_glib/meson.build +++ b/c_glib/meson.build @@ -32,7 +32,7 @@ project( # * 22.04: 0.61.2 # * 24.04: 1.3.2 meson_version: '>=0.61.2', - version: '26.0.0-SNAPSHOT', + version: '26.0.0', ) version = meson.project_version() diff --git a/c_glib/vcpkg.json b/c_glib/vcpkg.json index 8b16ab6c2009..77a423f77a5f 100644 --- a/c_glib/vcpkg.json +++ b/c_glib/vcpkg.json @@ -1,6 +1,6 @@ { "name": "arrow-glib", - "version-string": "26.0.0-SNAPSHOT", + "version-string": "26.0.0", "$comment:dependencies": "We can enable gobject-introspection again once it's updated", "dependencies": [ "glib", diff --git a/ci/scripts/PKGBUILD b/ci/scripts/PKGBUILD index e726372dae23..e9031a573613 100644 --- a/ci/scripts/PKGBUILD +++ b/ci/scripts/PKGBUILD @@ -18,7 +18,7 @@ _realname=arrow pkgbase=mingw-w64-${_realname} pkgname="${MINGW_PACKAGE_PREFIX}-${_realname}" -pkgver=25.0.1.9000 +pkgver=26.0.0 pkgrel=8000 pkgdesc="Apache Arrow is a cross-language development platform for in-memory data (mingw-w64)" arch=("any") diff --git a/ci/vcpkg/ports.patch b/ci/vcpkg/ports.patch index 8e62c75e7ac7..db38c25de3f8 100644 --- a/ci/vcpkg/ports.patch +++ b/ci/vcpkg/ports.patch @@ -80,3 +80,37 @@ index 278bc17a1c..d47d859360 100644 ) file(GLOB modules "${SOURCE_PATH}/cmake_modules/Find*.cmake") file(REMOVE ${modules} "${SOURCE_PATH}/c++/libs/libhdfspp/libhdfspp.tar.gz") +diff --git a/ports/uriparser/fix-glibc-2.28-reallocarray.diff b/ports/uriparser/fix-glibc-2.28-reallocarray.diff +new file mode 100644 +index 0000000..57adef4 +--- /dev/null ++++ b/ports/uriparser/fix-glibc-2.28-reallocarray.diff +@@ -0,0 +1,15 @@ ++diff --git a/src/UriMemory.c b/src/UriMemory.c ++--- a/src/UriMemory.c +++++ b/src/UriMemory.c ++@@ -50,6 +50,11 @@ ++ # define _DEFAULT_SOURCE 1 ++ # endif ++ +++// For glibc <2.29 +++# if !defined(_GNU_SOURCE) +++# define _GNU_SOURCE 1 +++# endif +++ ++ // For NetBSD (stdlib.h revision 1.122 of 2020-05-26) ++ # if defined(__NetBSD__) && !defined(_OPENBSD_SOURCE) ++ # define _OPENBSD_SOURCE 1 +diff --git a/ports/uriparser/portfile.cmake b/ports/uriparser/portfile.cmake +index 5df5e94..225e5fb 100644 +--- a/ports/uriparser/portfile.cmake ++++ b/ports/uriparser/portfile.cmake +@@ -4,6 +4,8 @@ vcpkg_from_github( + REF "uriparser-${VERSION}" + SHA512 1e4c397418e71e705b5712de2dc3ea6e95163c9d95bfbaa5f7d2fd2344d34eaf167d7d9389ab9671cfb314af026621b56e632ea34b681c2f4d7951a1b32a98b4 + HEAD_REF master ++ PATCHES ++ fix-glibc-2.28-reallocarray.diff + ) + + if("tool" IN_LIST FEATURES) diff --git a/cpp/CMakeLists.txt b/cpp/CMakeLists.txt index d39120dec2ed..0adf83b18199 100644 --- a/cpp/CMakeLists.txt +++ b/cpp/CMakeLists.txt @@ -96,7 +96,7 @@ if(POLICY CMP0170) cmake_policy(SET CMP0170 NEW) endif() -set(ARROW_VERSION "26.0.0-SNAPSHOT") +set(ARROW_VERSION "26.0.0") string(REGEX MATCH "^[0-9]+\\.[0-9]+\\.[0-9]+" ARROW_BASE_VERSION "${ARROW_VERSION}") diff --git a/cpp/examples/minimal_build/system_dependency.dockerfile b/cpp/examples/minimal_build/system_dependency.dockerfile index 773cd52baab4..ac7cef3496ea 100644 --- a/cpp/examples/minimal_build/system_dependency.dockerfile +++ b/cpp/examples/minimal_build/system_dependency.dockerfile @@ -34,6 +34,7 @@ RUN apt-get update -y -q && \ libre2-dev \ libsnappy-dev \ libthrift-dev \ + liburiparser-dev \ libutf8proc-dev \ libzstd-dev \ pkg-config \ diff --git a/cpp/meson.build b/cpp/meson.build index 5532d866db33..a2ba2675632a 100644 --- a/cpp/meson.build +++ b/cpp/meson.build @@ -19,7 +19,7 @@ project( 'arrow', 'cpp', 'c', - version: '26.0.0-SNAPSHOT', + version: '26.0.0', license: 'Apache-2.0', meson_version: '>=1.3.0', default_options: ['c_std=c11', 'warning_level=2', 'cpp_std=c++20'], diff --git a/cpp/src/arrow/util/async_generator_test.cc b/cpp/src/arrow/util/async_generator_test.cc index 9d965dbc4e5f..e51410b73e81 100644 --- a/cpp/src/arrow/util/async_generator_test.cc +++ b/cpp/src/arrow/util/async_generator_test.cc @@ -976,6 +976,9 @@ TEST_F(MergedGeneratorErrorHookTest, OuterErrorToWaiterNotOvertakenByLaterPull) // receives the error has completed, even when it is requested while the generator is // already completing. TEST_F(MergedGeneratorErrorHookTest, InnerErrorToWaiterNotOvertakenDuringCompletion) { +#ifndef ARROW_ENABLE_THREADING + GTEST_SKIP() << "Test requires threading support"; +#endif auto failing = Future::Make(); AsyncGenerator failing_sub = [failing]() { return failing; }; std::vector> subs = {failing_sub}; @@ -994,6 +997,9 @@ TEST_F(MergedGeneratorErrorHookTest, InnerErrorToWaiterNotOvertakenDuringComplet } TEST_F(MergedGeneratorErrorHookTest, OuterErrorToWaiterNotOvertakenDuringCompletion) { +#ifndef ARROW_ENABLE_THREADING + GTEST_SKIP() << "Test requires threading support"; +#endif auto failing = Future>::Make(); AsyncGenerator> source = [failing]() { return failing; }; MergedGenerator gen(std::move(source), 1); @@ -1011,6 +1017,9 @@ TEST_F(MergedGeneratorErrorHookTest, OuterErrorToWaiterNotOvertakenDuringComplet } TEST_F(MergedGeneratorErrorHookTest, ClaimedErrorNotOvertakenDuringCompletion) { +#ifndef ARROW_ENABLE_THREADING + GTEST_SKIP() << "Test requires threading support"; +#endif auto failing = Future::Make(); AsyncGenerator failing_sub = [failing]() { return failing; }; std::vector> subs = {MakeVectorGenerator({TestInt(1)}), diff --git a/cpp/src/parquet/reader_test.cc b/cpp/src/parquet/reader_test.cc index e4d2fd1d5900..b5487fa46899 100644 --- a/cpp/src/parquet/reader_test.cc +++ b/cpp/src/parquet/reader_test.cc @@ -1124,6 +1124,11 @@ class TestJSONWithLocalFile : public ::testing::Test { }; TEST_F(TestJSONWithLocalFile, JSONOutputWithStatistics) { + // simdjson 5.0 changed the format of fractured_json + if constexpr (simdjson::SIMDJSON_VERSION_MAJOR < 5) { + GTEST_SKIP() << "Test requires simdjson >= 5"; + } + std::string json_output = R"###({ "FileName": "nested_lists.snappy.parquet", "Version": "1.0", @@ -1138,15 +1143,9 @@ TEST_F(TestJSONWithLocalFile, JSONOutputWithStatistics) { "Name": "a.list.element.list.element.list.element", "PhysicalType": "BYTE_ARRAY", "ConvertedType": "UTF8", - "LogicalType": { "Type": "String" } + "LogicalType": {"Type": "String"} }, - { - "Id": "1", - "Name": "b", - "PhysicalType": "INT32", - "ConvertedType": "NONE", - "LogicalType": { "Type": "None" } - } + { "Id": "1", "Name": "b", "PhysicalType": "INT32", "ConvertedType": "NONE", "LogicalType": {"Type": "None"} } ], "RowGroups": [ { @@ -1191,6 +1190,11 @@ TEST_F(TestJSONWithLocalFile, JSONOutputWithStatistics) { } TEST_F(TestJSONWithLocalFile, JSONOutput) { + // simdjson 5.0 changed the format of fractured_json + if constexpr (simdjson::SIMDJSON_VERSION_MAJOR < 5) { + GTEST_SKIP() << "Test requires simdjson >= 5"; + } + std::string json_output = R"###({ "FileName": "alltypes_plain.parquet", "Version": "1.0", @@ -1200,17 +1204,77 @@ TEST_F(TestJSONWithLocalFile, JSONOutput) { "NumberOfRealColumns": "11", "NumberOfColumns": "11", "Columns": [ - { "ConvertedType": "NONE", "Id": "0" , "LogicalType": { "Type": "None" }, "Name": "id" , "PhysicalType": "INT32" }, - { "ConvertedType": "NONE", "Id": "1" , "LogicalType": { "Type": "None" }, "Name": "bool_col" , "PhysicalType": "BOOLEAN" }, - { "ConvertedType": "NONE", "Id": "2" , "LogicalType": { "Type": "None" }, "Name": "tinyint_col" , "PhysicalType": "INT32" }, - { "ConvertedType": "NONE", "Id": "3" , "LogicalType": { "Type": "None" }, "Name": "smallint_col" , "PhysicalType": "INT32" }, - { "ConvertedType": "NONE", "Id": "4" , "LogicalType": { "Type": "None" }, "Name": "int_col" , "PhysicalType": "INT32" }, - { "ConvertedType": "NONE", "Id": "5" , "LogicalType": { "Type": "None" }, "Name": "bigint_col" , "PhysicalType": "INT64" }, - { "ConvertedType": "NONE", "Id": "6" , "LogicalType": { "Type": "None" }, "Name": "float_col" , "PhysicalType": "FLOAT" }, - { "ConvertedType": "NONE", "Id": "7" , "LogicalType": { "Type": "None" }, "Name": "double_col" , "PhysicalType": "DOUBLE" }, - { "ConvertedType": "NONE", "Id": "8" , "LogicalType": { "Type": "None" }, "Name": "date_string_col", "PhysicalType": "BYTE_ARRAY" }, - { "ConvertedType": "NONE", "Id": "9" , "LogicalType": { "Type": "None" }, "Name": "string_col" , "PhysicalType": "BYTE_ARRAY" }, - { "ConvertedType": "NONE", "Id": "10", "LogicalType": { "Type": "None" }, "Name": "timestamp_col" , "PhysicalType": "INT96" } + { "Id": "0", "Name": "id", "PhysicalType": "INT32", "ConvertedType": "NONE", "LogicalType": {"Type": "None"} }, + { + "Id": "1", + "Name": "bool_col", + "PhysicalType": "BOOLEAN", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "2", + "Name": "tinyint_col", + "PhysicalType": "INT32", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "3", + "Name": "smallint_col", + "PhysicalType": "INT32", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "4", + "Name": "int_col", + "PhysicalType": "INT32", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "5", + "Name": "bigint_col", + "PhysicalType": "INT64", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "6", + "Name": "float_col", + "PhysicalType": "FLOAT", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "7", + "Name": "double_col", + "PhysicalType": "DOUBLE", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "8", + "Name": "date_string_col", + "PhysicalType": "BYTE_ARRAY", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "9", + "Name": "string_col", + "PhysicalType": "BYTE_ARRAY", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + }, + { + "Id": "10", + "Name": "timestamp_col", + "PhysicalType": "INT96", + "ConvertedType": "NONE", + "LogicalType": {"Type": "None"} + } ], "RowGroups": [ { @@ -1219,17 +1283,105 @@ TEST_F(TestJSONWithLocalFile, JSONOutput) { "TotalCompressedBytes": "0", "Rows": "8", "ColumnChunks": [ - { "CompressedSize": "73" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "0" , "StatsSet": "False", "UncompressedSize": "73" , "Values": "8" }, - { "CompressedSize": "24" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "1" , "StatsSet": "False", "UncompressedSize": "24" , "Values": "8" }, - { "CompressedSize": "47" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "2" , "StatsSet": "False", "UncompressedSize": "47" , "Values": "8" }, - { "CompressedSize": "47" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "3" , "StatsSet": "False", "UncompressedSize": "47" , "Values": "8" }, - { "CompressedSize": "47" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "4" , "StatsSet": "False", "UncompressedSize": "47" , "Values": "8" }, - { "CompressedSize": "55" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "5" , "StatsSet": "False", "UncompressedSize": "55" , "Values": "8" }, - { "CompressedSize": "47" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "6" , "StatsSet": "False", "UncompressedSize": "47" , "Values": "8" }, - { "CompressedSize": "55" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "7" , "StatsSet": "False", "UncompressedSize": "55" , "Values": "8" }, - { "CompressedSize": "88" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "8" , "StatsSet": "False", "UncompressedSize": "88" , "Values": "8" }, - { "CompressedSize": "49" , "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "9" , "StatsSet": "False", "UncompressedSize": "49" , "Values": "8" }, - { "CompressedSize": "139", "Compression": "UNCOMPRESSED", "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", "Id": "10", "StatsSet": "False", "UncompressedSize": "139", "Values": "8" } + { + "Id": "0", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "73", + "CompressedSize": "73" + }, + { + "Id": "1", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "24", + "CompressedSize": "24" + }, + { + "Id": "2", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "47", + "CompressedSize": "47" + }, + { + "Id": "3", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "47", + "CompressedSize": "47" + }, + { + "Id": "4", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "47", + "CompressedSize": "47" + }, + { + "Id": "5", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "55", + "CompressedSize": "55" + }, + { + "Id": "6", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "47", + "CompressedSize": "47" + }, + { + "Id": "7", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "55", + "CompressedSize": "55" + }, + { + "Id": "8", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "88", + "CompressedSize": "88" + }, + { + "Id": "9", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "49", + "CompressedSize": "49" + }, + { + "Id": "10", + "Values": "8", + "StatsSet": "False", + "Compression": "UNCOMPRESSED", + "Encodings": "RLE PLAIN_DICTIONARY PLAIN ", + "UncompressedSize": "139", + "CompressedSize": "139" + } ] } ] @@ -1243,6 +1395,11 @@ TEST_F(TestJSONWithLocalFile, JSONOutput) { TEST_F(TestJSONWithLocalFile, JSONOutputFLBA) { // min-max stats for FLBA contains non-utf8 output, so we don't check // the whole json output. + // simdjson 5.0 changed the format of fractured_json + if constexpr (simdjson::SIMDJSON_VERSION_MAJOR < 5) { + GTEST_SKIP() << "Test requires simdjson >= 5"; + } + std::string json_content = ReadFromLocalFile("fixed_length_byte_array.parquet"); std::string json_contains = R"###({ @@ -1259,7 +1416,7 @@ TEST_F(TestJSONWithLocalFile, JSONOutputFLBA) { "Name": "flba_field", "PhysicalType": "FIXED_LEN_BYTE_ARRAY(4)", "ConvertedType": "NONE", - "LogicalType": { "Type": "None" } + "LogicalType": {"Type": "None"} } ],)###"; @@ -1267,11 +1424,18 @@ TEST_F(TestJSONWithLocalFile, JSONOutputFLBA) { } TEST_F(TestJSONWithLocalFile, JSONOutputSortColumns) { + // simdjson 5.0 changed the format of fractured_json + if constexpr (simdjson::SIMDJSON_VERSION_MAJOR < 5) { + GTEST_SKIP() << "Test requires simdjson >= 5"; + } + std::string json_content = ReadFromLocalFile("sort_columns.parquet"); std::string json_contains = R"###("SortColumns": [ - { "column_idx": 0, "descending": 1, "nulls_first": 1 }, { "column_idx": 1, "descending": 0, "nulls_first": 0 } + {"column_idx": 0, "descending": 1, "nulls_first": 1}, + {"column_idx": 1, "descending": 0, "nulls_first": 0} ],)###"; + EXPECT_THAT(json_content, testing::HasSubstr(json_contains)); } diff --git a/cpp/vcpkg.json b/cpp/vcpkg.json index 4d35eb29d219..f94602e9ccde 100644 --- a/cpp/vcpkg.json +++ b/cpp/vcpkg.json @@ -1,6 +1,6 @@ { "name": "arrow", - "version-string": "26.0.0-SNAPSHOT", + "version-string": "26.0.0", "dependencies": [ "abseil", { diff --git a/dev/tasks/homebrew-formulae/apache-arrow-glib.rb b/dev/tasks/homebrew-formulae/apache-arrow-glib.rb index 469c662dfe85..c1b10e028ce1 100644 --- a/dev/tasks/homebrew-formulae/apache-arrow-glib.rb +++ b/dev/tasks/homebrew-formulae/apache-arrow-glib.rb @@ -29,7 +29,7 @@ class ApacheArrowGlib < Formula desc "GLib bindings for Apache Arrow" homepage "https://arrow.apache.org/" - url "https://www.apache.org/dyn/closer.lua?path=arrow/arrow-26.0.0-SNAPSHOT/apache-arrow-26.0.0-SNAPSHOT.tar.gz" + url "https://www.apache.org/dyn/closer.lua?path=arrow/arrow-26.0.0/apache-arrow-26.0.0.tar.gz" sha256 "9948ddb6d4798b51552d0dca3252dd6e3a7d0f9702714fc6f5a1b59397ce1d28" license "Apache-2.0" head "https://github.com/apache/arrow.git", branch: "main" diff --git a/dev/tasks/homebrew-formulae/apache-arrow.rb b/dev/tasks/homebrew-formulae/apache-arrow.rb index d2b99fd47573..4d64b7f6d5b7 100644 --- a/dev/tasks/homebrew-formulae/apache-arrow.rb +++ b/dev/tasks/homebrew-formulae/apache-arrow.rb @@ -29,7 +29,7 @@ class ApacheArrow < Formula desc "Columnar in-memory analytics layer designed to accelerate big data" homepage "https://arrow.apache.org/" - url "https://www.apache.org/dyn/closer.lua?path=arrow/arrow-26.0.0-SNAPSHOT/apache-arrow-26.0.0-SNAPSHOT.tar.gz" + url "https://www.apache.org/dyn/closer.lua?path=arrow/arrow-26.0.0/apache-arrow-26.0.0.tar.gz" sha256 "9948ddb6d4798b51552d0dca3252dd6e3a7d0f9702714fc6f5a1b59397ce1d28" license "Apache-2.0" head "https://github.com/apache/arrow.git", branch: "main" diff --git a/dev/tasks/linux-packages/apache-arrow-apt-source/debian/changelog b/dev/tasks/linux-packages/apache-arrow-apt-source/debian/changelog index 78a47e682fc4..9f3b72a2cbe7 100644 --- a/dev/tasks/linux-packages/apache-arrow-apt-source/debian/changelog +++ b/dev/tasks/linux-packages/apache-arrow-apt-source/debian/changelog @@ -1,3 +1,9 @@ +apache-arrow-apt-source (26.0.0-1) unstable; urgency=low + + * New upstream release. + + -- Raúl Cumplido Mon, 05 Oct 2026 08:18:39 -0000 + apache-arrow-apt-source (25.0.1-1) unstable; urgency=low * New upstream release. diff --git a/dev/tasks/linux-packages/apache-arrow-release/yum/apache-arrow-release.spec.in b/dev/tasks/linux-packages/apache-arrow-release/yum/apache-arrow-release.spec.in index 137f2dbbe84e..c0bde94449d1 100644 --- a/dev/tasks/linux-packages/apache-arrow-release/yum/apache-arrow-release.spec.in +++ b/dev/tasks/linux-packages/apache-arrow-release/yum/apache-arrow-release.spec.in @@ -85,6 +85,9 @@ else fi %changelog +* Mon Oct 05 2026 Raúl Cumplido - 26.0.0-1 +- New upstream release. + * Wed Aug 05 2026 Raúl Cumplido - 25.0.1-1 - New upstream release. diff --git a/dev/tasks/linux-packages/apache-arrow/debian/changelog b/dev/tasks/linux-packages/apache-arrow/debian/changelog index 71ccc78187af..a9ed3ccd9fdc 100644 --- a/dev/tasks/linux-packages/apache-arrow/debian/changelog +++ b/dev/tasks/linux-packages/apache-arrow/debian/changelog @@ -1,3 +1,9 @@ +apache-arrow (26.0.0-1) unstable; urgency=low + + * New upstream release. + + -- Raúl Cumplido Mon, 05 Oct 2026 08:18:39 -0000 + apache-arrow (25.0.1-1) unstable; urgency=low * New upstream release. diff --git a/dev/tasks/linux-packages/apache-arrow/yum/arrow.spec.in b/dev/tasks/linux-packages/apache-arrow/yum/arrow.spec.in index 40e8862a5b97..ef3e9d5a4f38 100644 --- a/dev/tasks/linux-packages/apache-arrow/yum/arrow.spec.in +++ b/dev/tasks/linux-packages/apache-arrow/yum/arrow.spec.in @@ -942,6 +942,9 @@ Documentation for Apache Parquet GLib. %endif %changelog +* Mon Oct 05 2026 Raúl Cumplido - 26.0.0-1 +- New upstream release. + * Wed Aug 05 2026 Raúl Cumplido - 25.0.1-1 - New upstream release. diff --git a/docs/source/_static/versions.json b/docs/source/_static/versions.json index c7e9aa3bf3de..ee496b3900f7 100644 --- a/docs/source/_static/versions.json +++ b/docs/source/_static/versions.json @@ -1,15 +1,20 @@ [ { - "name": "26.0 (dev)", + "name": "27.0 (dev)", "version": "dev/", "url": "https://arrow.apache.org/docs/dev/" }, { - "name": "25.0 (stable)", + "name": "26.0 (stable)", "version": "", "url": "https://arrow.apache.org/docs/", "preferred": true }, + { + "name": "25.0", + "version": "25.0/", + "url": "https://arrow.apache.org/docs/25.0/" + }, { "name": "24.0", "version": "24.0/", diff --git a/matlab/CMakeLists.txt b/matlab/CMakeLists.txt index 490a63bccda1..44d092aac477 100644 --- a/matlab/CMakeLists.txt +++ b/matlab/CMakeLists.txt @@ -100,7 +100,7 @@ endfunction() set(CMAKE_CXX_STANDARD 20) -set(MLARROW_VERSION "26.0.0-SNAPSHOT") +set(MLARROW_VERSION "26.0.0") string(REGEX MATCH "^[0-9]+\\.[0-9]+\\.[0-9]+" MLARROW_BASE_VERSION "${MLARROW_VERSION}") project(mlarrow VERSION "${MLARROW_BASE_VERSION}") diff --git a/python/CMakeLists.txt b/python/CMakeLists.txt index 7f6f1aebeef8..330bb2b6399a 100644 --- a/python/CMakeLists.txt +++ b/python/CMakeLists.txt @@ -28,7 +28,7 @@ project(pyarrow) # which in turn meant that Py_GIL_DISABLED was not set. set(CMAKE_NO_SYSTEM_FROM_IMPORTED ON) -set(PYARROW_VERSION "26.0.0-SNAPSHOT") +set(PYARROW_VERSION "26.0.0") string(REGEX MATCH "^[0-9]+\\.[0-9]+\\.[0-9]+" PYARROW_BASE_VERSION "${PYARROW_VERSION}") # Generate SO version and full SO version diff --git a/python/pyproject.toml b/python/pyproject.toml index 68c1a807dd51..22db9b8492a2 100644 --- a/python/pyproject.toml +++ b/python/pyproject.toml @@ -109,7 +109,7 @@ root = '..' version_file = 'pyarrow/_generated_version.py' version_scheme = 'guess-next-dev' git_describe_command = 'git describe --dirty --tags --long --match "apache-arrow-[0-9]*.*"' -fallback_version = '26.0.0a0' +fallback_version = '26.0.0' # TODO: Enable type checking once stubs are merged [tool.mypy] diff --git a/r/DESCRIPTION b/r/DESCRIPTION index 43f489f490ed..c3e688cb0dc9 100644 --- a/r/DESCRIPTION +++ b/r/DESCRIPTION @@ -1,6 +1,6 @@ Package: arrow Title: Integration to 'Apache' 'Arrow' -Version: 25.0.1.9000 +Version: 26.0.0 Authors@R: c( person("Neal", "Richardson", email = "neal.p.richardson@gmail.com", role = c("aut")), person("Ian", "Cook", email = "ianmcook@gmail.com", role = c("aut")), diff --git a/r/NEWS.md b/r/NEWS.md index 1d1466de377d..bfd75d5991b5 100644 --- a/r/NEWS.md +++ b/r/NEWS.md @@ -17,7 +17,7 @@ under the License. --> -# arrow 25.0.1.9000 +# arrow 26.0.0 ## Breaking changes diff --git a/r/pkgdown/assets/versions.html b/r/pkgdown/assets/versions.html index bc08cbdf388f..23e81276d807 100644 --- a/r/pkgdown/assets/versions.html +++ b/r/pkgdown/assets/versions.html @@ -1,7 +1,8 @@ -

25.0.1.9000 (dev)

-

25.0.1 (release)

+

26.0.0.9000 (dev)

+

26.0.0 (release)

+

25.0.1

24.0.0

23.0.1

22.0.0

diff --git a/r/pkgdown/assets/versions.json b/r/pkgdown/assets/versions.json index cd18e93044dd..43039b7b193c 100644 --- a/r/pkgdown/assets/versions.json +++ b/r/pkgdown/assets/versions.json @@ -1,12 +1,16 @@ [ { - "name": "25.0.1.9000 (dev)", + "name": "26.0.0.9000 (dev)", "version": "dev/" }, { - "name": "25.0.1 (release)", + "name": "26.0.0 (release)", "version": "" }, + { + "name": "25.0.1", + "version": "25.0/" + }, { "name": "24.0.0", "version": "24.0/" diff --git a/ruby/red-arrow-cuda/lib/arrow-cuda/version.rb b/ruby/red-arrow-cuda/lib/arrow-cuda/version.rb index 0d93e5966b26..2787743e0c74 100644 --- a/ruby/red-arrow-cuda/lib/arrow-cuda/version.rb +++ b/ruby/red-arrow-cuda/lib/arrow-cuda/version.rb @@ -16,7 +16,7 @@ # under the License. module ArrowCUDA - VERSION = "26.0.0-SNAPSHOT" + VERSION = "26.0.0" module Version numbers, TAG = VERSION.split("-") diff --git a/ruby/red-arrow-dataset/lib/arrow-dataset/version.rb b/ruby/red-arrow-dataset/lib/arrow-dataset/version.rb index a553cdda6cc4..f9c9ff03b77f 100644 --- a/ruby/red-arrow-dataset/lib/arrow-dataset/version.rb +++ b/ruby/red-arrow-dataset/lib/arrow-dataset/version.rb @@ -16,7 +16,7 @@ # under the License. module ArrowDataset - VERSION = "26.0.0-SNAPSHOT" + VERSION = "26.0.0" module Version numbers, TAG = VERSION.split("-") diff --git a/ruby/red-arrow-flight-sql/lib/arrow-flight-sql/version.rb b/ruby/red-arrow-flight-sql/lib/arrow-flight-sql/version.rb index 97f7061491ce..d93daeda9fcd 100644 --- a/ruby/red-arrow-flight-sql/lib/arrow-flight-sql/version.rb +++ b/ruby/red-arrow-flight-sql/lib/arrow-flight-sql/version.rb @@ -16,7 +16,7 @@ # under the License. module ArrowFlightSQL - VERSION = "26.0.0-SNAPSHOT" + VERSION = "26.0.0" module Version numbers, TAG = VERSION.split("-") diff --git a/ruby/red-arrow-flight/lib/arrow-flight/version.rb b/ruby/red-arrow-flight/lib/arrow-flight/version.rb index 0a7eeb27a703..9b413361c6a4 100644 --- a/ruby/red-arrow-flight/lib/arrow-flight/version.rb +++ b/ruby/red-arrow-flight/lib/arrow-flight/version.rb @@ -16,7 +16,7 @@ # under the License. module ArrowFlight - VERSION = "26.0.0-SNAPSHOT" + VERSION = "26.0.0" module Version numbers, TAG = VERSION.split("-") diff --git a/ruby/red-arrow-format/lib/arrow-format/version.rb b/ruby/red-arrow-format/lib/arrow-format/version.rb index ee45b1627d49..ec5580ca18ae 100644 --- a/ruby/red-arrow-format/lib/arrow-format/version.rb +++ b/ruby/red-arrow-format/lib/arrow-format/version.rb @@ -16,7 +16,7 @@ # under the License. module ArrowFormat - VERSION = "26.0.0-SNAPSHOT" + VERSION = "26.0.0" module Version numbers, TAG = VERSION.split("-") diff --git a/ruby/red-arrow/lib/arrow/version.rb b/ruby/red-arrow/lib/arrow/version.rb index c8cacd0bffe7..669f5d4084f1 100644 --- a/ruby/red-arrow/lib/arrow/version.rb +++ b/ruby/red-arrow/lib/arrow/version.rb @@ -16,7 +16,7 @@ # under the License. module Arrow - VERSION = "26.0.0-SNAPSHOT" + VERSION = "26.0.0" module Version numbers, TAG = VERSION.split("-") diff --git a/ruby/red-gandiva/lib/gandiva/version.rb b/ruby/red-gandiva/lib/gandiva/version.rb index 36bf8531ae4f..e59449fd2ebd 100644 --- a/ruby/red-gandiva/lib/gandiva/version.rb +++ b/ruby/red-gandiva/lib/gandiva/version.rb @@ -16,7 +16,7 @@ # under the License. module Gandiva - VERSION = "26.0.0-SNAPSHOT" + VERSION = "26.0.0" module Version numbers, TAG = VERSION.split("-") diff --git a/ruby/red-parquet/lib/parquet/version.rb b/ruby/red-parquet/lib/parquet/version.rb index c22e97135d9b..ca8e39ec71df 100644 --- a/ruby/red-parquet/lib/parquet/version.rb +++ b/ruby/red-parquet/lib/parquet/version.rb @@ -16,7 +16,7 @@ # under the License. module Parquet - VERSION = "26.0.0-SNAPSHOT" + VERSION = "26.0.0" module Version numbers, TAG = VERSION.split("-")