This is an automated email from the ASF dual-hosted git repository.
thisisnic pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow.git
The following commit(s) were added to refs/heads/main by this push:
new 26310ff4ad2 GH-51655: [R] Advance long-standing deprecations in
map_batches() and pull() (#51658)
26310ff4ad2 is described below
commit 26310ff4ad294ca4477597d96de8c752e77ccdfa
Author: Nic Crane <[email protected]>
AuthorDate: Wed Sep 30 14:41:34 2026 +0100
GH-51655: [R] Advance long-standing deprecations in map_batches() and
pull() (#51658)
### Rationale for this change
Old deprecation warnings hadn't been dealt with
### What changes are included in this PR?
I've deprecated the `.data.frame` argument in `map_batches()` but I think
we actually may be better off leaving `pull()` as-is given it's been like that
for such a long time.
### Are these changes tested?
Existing tests
### Are there any user-facing changes?
Yes
`pull()` on Arrow data no longer warns about a future change of default; it
will keep returning an R vector by default, with `as_vector = FALSE` or
`options(arrow.pull_as_vector = FALSE)` to get a `ChunkedArray`.
The deprecated `.data.frame` argument to `map_batches()` has been removed.
### Was AI used for this PR?
In accordance to the [AI generation
guidelines](https://arrow.apache.org/docs/dev/developers/overview.html#ai-generated-code),
please disclose below whether and how AI was used in this PR.
**PR code and description written by:**
- [x] Human
- [x] AI
**Reviewed before submission by:**
- [x] Human
- [ ] AI
- [ ] Not reviewed
* GitHub Issue: #51655
Authored-by: Nic Crane <[email protected]>
Signed-off-by: Nic Crane <[email protected]>
---
r/NEWS.md | 10 ++++++++++
r/R/arrow-package.R | 7 +++----
r/R/dataset-scan.R | 10 +---------
r/R/dplyr-collect.R | 24 ++----------------------
r/R/dplyr-funcs-doc.R | 2 +-
r/man/acero.Rd | 2 +-
r/man/map_batches.Rd | 4 +---
r/tests/testthat/helper-arrow.R | 4 ----
r/tests/testthat/test-dplyr-query.R | 26 +++++++++++++++-----------
9 files changed, 34 insertions(+), 55 deletions(-)
diff --git a/r/NEWS.md b/r/NEWS.md
index 185316beb8e..1bda314ee7f 100644
--- a/r/NEWS.md
+++ b/r/NEWS.md
@@ -19,6 +19,11 @@
# arrow 25.0.1.9000
+## Breaking changes
+
+- The `.data.frame` argument to `map_batches()`, deprecated since 9.0.0, has
+ been removed. Call `collect()` on the result to get a data frame (#51655).
+
## Minor improvements and fixes
- Factor levels inside list columns are now unified across the whole column
@@ -26,6 +31,11 @@
`read_ipc_stream()` or `open_dataset()`) produces valid factors that can be
unnested. Similarly, `int64` and `uint32` values inside list columns are
converted to a single R type across the column (#50514).
+- `pull()` on Arrow data no longer warns about a future change of default.
+ The planned switch to returning a `ChunkedArray` has been dropped, so
+ `pull()` will keep returning an R vector by default; use
+ `as_vector = FALSE` or `options(arrow.pull_as_vector = FALSE)` to get a
+ `ChunkedArray` (#51655).
# arrow 25.0.1
diff --git a/r/R/arrow-package.R b/r/R/arrow-package.R
index 2706faee5cb..5e4524f2f77 100644
--- a/r/R/arrow-package.R
+++ b/r/R/arrow-package.R
@@ -56,10 +56,9 @@ supported_dplyr_methods <- list(
rename = NULL,
pull = c(
"the `name` argument is not supported;",
- "returns an R vector by default but this behavior is deprecated and will",
- "return an Arrow [ChunkedArray] in a future release. Provide",
- "`as_vector = TRUE/FALSE` to control this behavior, or set",
- "`options(arrow.pull_as_vector)` globally."
+ "returns an R vector by default. Provide `as_vector = FALSE`",
+ "to return an Arrow [ChunkedArray] instead, or set",
+ "`options(arrow.pull_as_vector = FALSE)` globally."
),
relocate = NULL,
compute = NULL,
diff --git a/r/R/dataset-scan.R b/r/R/dataset-scan.R
index c4b651d419d..68b59cdb724 100644
--- a/r/R/dataset-scan.R
+++ b/r/R/dataset-scan.R
@@ -220,17 +220,9 @@ tail_from_batches <- function(batches, n) {
#' the result; use `FALSE` to evaluate `FUN` on all batches before returning
#' the reader.
#' @param ... Additional arguments passed to `FUN`
-#' @param .data.frame Deprecated argument, ignored
#' @return An `arrow_dplyr_query`.
#' @export
-map_batches <- function(X, FUN, ..., .schema = NULL, .lazy = TRUE, .data.frame
= NULL) {
- if (!is.null(.data.frame)) {
- warning(
- "The .data.frame argument is deprecated. ",
- "Call collect() on the result to get a data.frame.",
- call. = FALSE
- )
- }
+map_batches <- function(X, FUN, ..., .schema = NULL, .lazy = TRUE) {
FUN <- as_mapper(FUN)
reader <- as_record_batch_reader(X)
dots <- list2(...)
diff --git a/r/R/dplyr-collect.R b/r/R/dplyr-collect.R
index b9461df20d3..2b415e22ca9 100644
--- a/r/R/dplyr-collect.R
+++ b/r/R/dplyr-collect.R
@@ -56,7 +56,7 @@ compute.arrow_dplyr_query <- function(x, ...) {
}
compute.Dataset <- compute.RecordBatchReader <- compute.arrow_dplyr_query
-pull.Dataset <- function(.data, var = -1, ..., as_vector =
getOption("arrow.pull_as_vector")) {
+pull.Dataset <- function(.data, var = -1, ..., as_vector =
getOption("arrow.pull_as_vector", TRUE)) {
.data <- as_adq(.data)
var <- vars_pull(names(.data), !!enquo(var))
.data$selected_columns <- set_names(.data$selected_columns[var], var)
@@ -65,32 +65,12 @@ pull.Dataset <- function(.data, var = -1, ..., as_vector =
getOption("arrow.pull
}
pull.RecordBatchReader <- pull.arrow_dplyr_query <- pull.Dataset
-pull.ArrowTabular <- function(x, var = -1, ..., as_vector =
getOption("arrow.pull_as_vector")) {
+pull.ArrowTabular <- function(x, var = -1, ..., as_vector =
getOption("arrow.pull_as_vector", TRUE)) {
out <- x[[vars_pull(names(x), !!enquo(var))]]
handle_pull_as_vector(out, as_vector)
}
handle_pull_as_vector <- function(out, as_vector) {
- if (is.null(as_vector)) {
- warn(
- c(
- paste(
- "Default behavior of `pull()` on Arrow data is changing. Current",
- "behavior of returning an R vector is deprecated, and in a future",
- "release, it will return an Arrow `ChunkedArray`. To control this:"
- ),
- i = paste(
- "Specify `as_vector = TRUE` (the current default) or",
- "`FALSE` (what it will change to) in `pull()`"
- ),
- i = "Or, set `options(arrow.pull_as_vector)` globally"
- ),
- .frequency = "regularly",
- .frequency_id = "arrow.pull_as_vector",
- class = "lifecycle_warning_deprecated"
- )
- as_vector <- TRUE
- }
if (as_vector) {
out <- as.vector(out)
}
diff --git a/r/R/dplyr-funcs-doc.R b/r/R/dplyr-funcs-doc.R
index 61dbf618d32..bbcd3fe59f4 100644
--- a/r/R/dplyr-funcs-doc.R
+++ b/r/R/dplyr-funcs-doc.R
@@ -55,7 +55,7 @@
#' * [`inner_join()`][dplyr::inner_join()]: the `copy` argument is ignored
#' * [`left_join()`][dplyr::left_join()]: the `copy` argument is ignored
#' * [`mutate()`][dplyr::mutate()]
-#' * [`pull()`][dplyr::pull()]: the `name` argument is not supported; returns
an R vector by default but this behavior is deprecated and will return an Arrow
[ChunkedArray] in a future release. Provide `as_vector = TRUE/FALSE` to control
this behavior, or set `options(arrow.pull_as_vector)` globally.
+#' * [`pull()`][dplyr::pull()]: the `name` argument is not supported; returns
an R vector by default. Provide `as_vector = FALSE` to return an Arrow
[ChunkedArray] instead, or set `options(arrow.pull_as_vector = FALSE)` globally.
#' * [`relocate()`][dplyr::relocate()]
#' * [`rename()`][dplyr::rename()]
#' * [`rename_with()`][dplyr::rename_with()]
diff --git a/r/man/acero.Rd b/r/man/acero.Rd
index 1203fe4c43a..45be81521e8 100644
--- a/r/man/acero.Rd
+++ b/r/man/acero.Rd
@@ -42,7 +42,7 @@ Table into an R \code{tibble}.
\item \code{\link[dplyr:inner_join]{inner_join()}}: the \code{copy} argument
is ignored
\item \code{\link[dplyr:left_join]{left_join()}}: the \code{copy} argument is
ignored
\item \code{\link[dplyr:mutate]{mutate()}}
-\item \code{\link[dplyr:pull]{pull()}}: the \code{name} argument is not
supported; returns an R vector by default but this behavior is deprecated and
will return an Arrow \link{ChunkedArray} in a future release. Provide
\code{as_vector = TRUE/FALSE} to control this behavior, or set
\code{options(arrow.pull_as_vector)} globally.
+\item \code{\link[dplyr:pull]{pull()}}: the \code{name} argument is not
supported; returns an R vector by default. Provide \code{as_vector = FALSE} to
return an Arrow \link{ChunkedArray} instead, or set
\code{options(arrow.pull_as_vector = FALSE)} globally.
\item \code{\link[dplyr:relocate]{relocate()}}
\item \code{\link[dplyr:rename]{rename()}}
\item \code{\link[dplyr:rename_with]{rename_with()}}
diff --git a/r/man/map_batches.Rd b/r/man/map_batches.Rd
index a147e268a96..e8409359bb2 100644
--- a/r/man/map_batches.Rd
+++ b/r/man/map_batches.Rd
@@ -4,7 +4,7 @@
\alias{map_batches}
\title{Apply a function to a stream of RecordBatches}
\usage{
-map_batches(X, FUN, ..., .schema = NULL, .lazy = TRUE, .data.frame = NULL)
+map_batches(X, FUN, ..., .schema = NULL, .lazy = TRUE)
}
\arguments{
\item{X}{A \code{Dataset} or \code{arrow_dplyr_query} object, as returned by
the
@@ -22,8 +22,6 @@ from the first batch.}
\item{.lazy}{Use \code{TRUE} to evaluate \code{FUN} lazily as batches are read
from
the result; use \code{FALSE} to evaluate \code{FUN} on all batches before
returning
the reader.}
-
-\item{.data.frame}{Deprecated argument, ignored}
}
\value{
An \code{arrow_dplyr_query}.
diff --git a/r/tests/testthat/helper-arrow.R b/r/tests/testthat/helper-arrow.R
index fb22aac0adc..56f17872b16 100644
--- a/r/tests/testthat/helper-arrow.R
+++ b/r/tests/testthat/helper-arrow.R
@@ -29,10 +29,6 @@ Sys.setlocale("LC_COLLATE", "C")
# (R CMD check does this, but in case you're running outside of check)
Sys.setenv(LANGUAGE = "en")
-# Set this option so that the deprecation warning isn't shown
-# (except when we test for it)
-options(arrow.pull_as_vector = FALSE)
-
with_language <- function(lang, expr) {
skip_on_cran()
skip_if_not(capabilities("NLS"))
diff --git a/r/tests/testthat/test-dplyr-query.R
b/r/tests/testthat/test-dplyr-query.R
index 53864a16633..ac69da80d42 100644
--- a/r/tests/testthat/test-dplyr-query.R
+++ b/r/tests/testthat/test-dplyr-query.R
@@ -94,17 +94,6 @@ test_that("pull", {
)
})
-test_that("pull() shows a deprecation warning if the option isn't set", {
- expect_warning(
- vec <- tbl |>
- arrow_table() |>
- pull(as_vector = NULL),
- "Current behavior of returning an R vector is deprecated"
- )
- # And the default is the old behavior, an R vector
- expect_identical(vec, pull(tbl))
-})
-
test_that("collect(as_data_frame=FALSE)", {
batch <- record_batch(tbl)
@@ -761,3 +750,18 @@ test_that("nested field ref error handling", {
"No match"
)
})
+
+test_that("pull() returns an R vector by default", {
+ withr::local_options(arrow.pull_as_vector = NULL)
+ tab <- arrow_table(tbl)
+
+ expect_no_warning(out <- pull(tab, int))
+ expect_identical(out, tbl$int)
+ expect_no_warning(out <- pull(tab |> filter(int > 4), int))
+ expect_identical(out, tbl$int[!is.na(tbl$int) & tbl$int > 4])
+
+ expect_r6_class(pull(tab, int, as_vector = FALSE), "ChunkedArray")
+
+ withr::local_options(arrow.pull_as_vector = FALSE)
+ expect_r6_class(pull(tab, int), "ChunkedArray")
+})