This is an automated email from the ASF dual-hosted git repository.

thisisnic pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow.git


The following commit(s) were added to refs/heads/main by this push:
     new 1a04cfb98ab GH-38771: [R][Documentation] Document add_filename on 
open_dataset help page (#51392)
1a04cfb98ab is described below

commit 1a04cfb98ababb0eab44206bdd49dc88c7301196
Author: Nic Crane <[email protected]>
AuthorDate: Wed Sep 30 14:28:01 2026 +0100

    GH-38771: [R][Documentation] Document add_filename on open_dataset help 
page (#51392)
    
    ### Rationale for this change
    
    `add_filename()` not documented clearly
    
    ### What changes are included in this PR?
    
    Document it better
    
    ### Are these changes tested?
    
    No
    
    ### Are there any user-facing changes?
    
    No just docs
    
    ### Was AI used for this PR?
    
    In accordance to the [AI generation 
guidelines](https://arrow.apache.org/docs/dev/developers/overview.html#ai-generated-code),
 please disclose below whether and how AI was used in this PR.
    
    **PR code and description written by:**
    
    - [x] Human
    - [x] AI
    
    **Reviewed before submission by:**
    
    - [x] Human
    - [x] AI
    - [ ] Not reviewed
    
    * GitHub Issue: #38771
    
    Authored-by: Nic Crane <[email protected]>
    Signed-off-by: Nic Crane <[email protected]>
---
 r/R/dataset.R                   | 19 ++++++++++++++++++-
 r/R/dplyr-funcs-augmented.R     | 37 ++++++++++++++++++++++++++-----------
 r/R/util.R                      |  4 ++--
 r/_pkgdown.yml                  |  2 ++
 r/man/add_filename.Rd           | 38 +++++++++++++++++++++++++++-----------
 r/man/open_dataset.Rd           | 20 +++++++++++++++++++-
 r/tests/testthat/test-dataset.R |  4 ++--
 r/vignettes/dataset.Rmd         | 17 +++++++++++++++++
 8 files changed, 113 insertions(+), 28 deletions(-)

diff --git a/r/R/dataset.R b/r/R/dataset.R
index d58ea7d984d..7f28c207bee 100644
--- a/r/R/dataset.R
+++ b/r/R/dataset.R
@@ -66,6 +66,23 @@
 #' column types, as described above. If neither are provided, no partitioning
 #' information will be taken from the file paths.
 #'
+#' @section Adding the source filename as a column:
+#'
+#' Partitioning only recovers information encoded in directory names. If you
+#' need to know which file each row came from, call [add_filename()] inside a
+#' `dplyr` query on the dataset:
+#'
+#' ```r
+#' open_dataset("nyc-taxi") |>
+#'   mutate(file = add_filename()) |>
+#'   collect()
+#' ```
+#'
+#' This is useful, for example, when you have opened a subdirectory of a
+#' partitioned dataset directly (so the partition columns above that directory
+#' are not inferred) and want to recover the partition values from the path.
+#' See [add_filename()] for details and limitations.
+#'
 #' @param sources One of:
 #'   * a string path or URI to a directory containing data files
 #'   * a [FileSystem] that references a directory containing data files
@@ -127,7 +144,7 @@
 #' or call [`$NewScan()`][Scanner] to construct a query directly.
 #' @export
 #' @seealso \href{https://arrow.apache.org/docs/r/articles/dataset.html}{
-#' datasets article}
+#' datasets article}, [add_filename()]
 #' @include arrow-object.R
 #' @examplesIf arrow_with_dataset() & arrow_with_parquet()
 #' # Set up directory for examples
diff --git a/r/R/dplyr-funcs-augmented.R b/r/R/dplyr-funcs-augmented.R
index 97b924b5e11..0d47d7d729c 100644
--- a/r/R/dplyr-funcs-augmented.R
+++ b/r/R/dplyr-funcs-augmented.R
@@ -18,27 +18,42 @@
 #' Add the data filename as a column
 #'
 #' This function only exists inside `arrow` `dplyr` queries, and it only is
-#' valid when querying on a `FileSystemDataset`.
+#' valid when querying on a `FileSystemDataset`, such as one created by
+#' [open_dataset()]. Use it inside `mutate()` to add a column holding the path
+#' of the file each row was read from.
 #'
-#' To use filenames generated by this function in subsequent pipeline steps, 
you
-#' must either call \code{\link[dplyr:compute]{compute()}} or
-#' \code{\link[dplyr:collect]{collect()}} first. See Examples.
+#' The filename column can be used in later `select()`, `arrange()` and
+#' `group_by()` steps of the same query. However, it can't be used in
+#' `filter()`, and some functions (such as `substr()`) are not supported on it.
+#' In these cases, call \code{\link[dplyr:compute]{compute()}} or
+#' \code{\link[dplyr:collect]{collect()}} first. [add_filename()] must also be
+#' called before any aggregation or join. See Examples.
 #'
 #' @return A `FieldRef` \code{\link{Expression}} that refers to the filename
 #' augmented column.
 #'
+#' @seealso [open_dataset()]
+#'
 #' @examples \dontrun{
-#' open_dataset("nyc-taxi") |> mutate(
-#'   file =
-#'     add_filename()
-#' )
+#' open_dataset("nyc-taxi") |>
+#'   mutate(file = add_filename()) |>
+#'   collect()
+#'
+#' # Simple expressions on the new column work in a later mutate(), for
+#' # example to recover a partition value from the path
+#' open_dataset("nyc-taxi/year=2015") |>
+#'   mutate(file = add_filename()) |>
+#'   mutate(year_from_path = sub(".*year=([0-9]{4}).*", "\\1", file)) |>
+#'   collect()
 #'
-#' # To use a verb like mutate() with add_filename() we need to first call
-#' # compute()
+#' # To filter() on the new column, or use functions such as substr() that
+#' # need to know its type, call compute() or collect() first
 #' open_dataset("nyc-taxi") |>
 #'   mutate(file = add_filename()) |>
 #'   compute() |>
-#'   mutate(filename_length = nchar(file))
+#'   filter(endsWith(file, "part-0.parquet")) |>
+#'   mutate(file_start = substr(file, 1, 10)) |>
+#'   collect()
 #' }
 #'
 #' @keywords internal
diff --git a/r/R/util.R b/r/R/util.R
index dc48c5fb8ff..233b0cb6869 100644
--- a/r/R/util.R
+++ b/r/R/util.R
@@ -230,8 +230,8 @@ handle_augmented_field_misuse <- function(msg, call) {
       i = paste(
         "`add_filename()` or use of the `__filename` augmented field can only",
         "be used with Dataset objects, can only be added before doing",
-        "an aggregation or a join, and cannot be referenced in subsequent",
-        "pipeline steps until either compute() or collect() is called."
+        "an aggregation or a join, and cannot be used in filter() until",
+        "either compute() or collect() is called."
       )
     )
     abort(msg, call = call)
diff --git a/r/_pkgdown.yml b/r/_pkgdown.yml
index 5aa96dfff41..a9bc60ae49e 100644
--- a/r/_pkgdown.yml
+++ b/r/_pkgdown.yml
@@ -244,6 +244,8 @@ reference:
     Functionality for computing values on Arrow data objects.
   contents:
   - acero
+  - add_filename
+  - cast
   - arrow-functions
   - arrow-verbs
   - arrow-dplyr
diff --git a/r/man/add_filename.Rd b/r/man/add_filename.Rd
index 29457e25d38..22f659d8ab3 100644
--- a/r/man/add_filename.Rd
+++ b/r/man/add_filename.Rd
@@ -12,27 +12,43 @@ augmented column.
 }
 \description{
 This function only exists inside \code{arrow} \code{dplyr} queries, and it 
only is
-valid when querying on a \code{FileSystemDataset}.
+valid when querying on a \code{FileSystemDataset}, such as one created by
+\code{\link[=open_dataset]{open_dataset()}}. Use it inside \code{mutate()} to 
add a column holding the path
+of the file each row was read from.
 }
 \details{
-To use filenames generated by this function in subsequent pipeline steps, you
-must either call \code{\link[dplyr:compute]{compute()}} or
-\code{\link[dplyr:collect]{collect()}} first. See Examples.
+The filename column can be used in later \code{select()}, \code{arrange()} and
+\code{group_by()} steps of the same query. However, it can't be used in
+\code{filter()}, and some functions (such as \code{substr()}) are not 
supported on it.
+In these cases, call \code{\link[dplyr:compute]{compute()}} or
+\code{\link[dplyr:collect]{collect()}} first. 
\code{\link[=add_filename]{add_filename()}} must also be
+called before any aggregation or join. See Examples.
 }
 \examples{
 \dontrun{
-open_dataset("nyc-taxi") |> mutate(
-  file =
-    add_filename()
-)
+open_dataset("nyc-taxi") |>
+  mutate(file = add_filename()) |>
+  collect()
 
-# To use a verb like mutate() with add_filename() we need to first call
-# compute()
+# Simple expressions on the new column work in a later mutate(), for
+# example to recover a partition value from the path
+open_dataset("nyc-taxi/year=2015") |>
+  mutate(file = add_filename()) |>
+  mutate(year_from_path = sub(".*year=([0-9]{4}).*", "\\\\1", file)) |>
+  collect()
+
+# To filter() on the new column, or use functions such as substr() that
+# need to know its type, call compute() or collect() first
 open_dataset("nyc-taxi") |>
   mutate(file = add_filename()) |>
   compute() |>
-  mutate(filename_length = nchar(file))
+  filter(endsWith(file, "part-0.parquet")) |>
+  mutate(file_start = substr(file, 1, 10)) |>
+  collect()
 }
 
+}
+\seealso{
+\code{\link[=open_dataset]{open_dataset()}}
 }
 \keyword{internal}
diff --git a/r/man/open_dataset.Rd b/r/man/open_dataset.Rd
index 93ab25ed527..5f2c7268025 100644
--- a/r/man/open_dataset.Rd
+++ b/r/man/open_dataset.Rd
@@ -163,6 +163,24 @@ column types, as described above. If neither are provided, 
no partitioning
 information will be taken from the file paths.
 }
 
+\section{Adding the source filename as a column}{
+
+
+Partitioning only recovers information encoded in directory names. If you
+need to know which file each row came from, call 
\code{\link[=add_filename]{add_filename()}} inside a
+\code{dplyr} query on the dataset:
+
+\if{html}{\out{<div class="sourceCode 
r">}}\preformatted{open_dataset("nyc-taxi") |>
+  mutate(file = add_filename()) |>
+  collect()
+}\if{html}{\out{</div>}}
+
+This is useful, for example, when you have opened a subdirectory of a
+partitioned dataset directly (so the partition columns above that directory
+are not inferred) and want to recover the partition values from the path.
+See \code{\link[=add_filename]{add_filename()}} for details and limitations.
+}
+
 \examples{
 \dontshow{if (arrow_with_dataset() & arrow_with_parquet()) withAutoprint(\{ # 
examplesIf}
 # Set up directory for examples
@@ -219,5 +237,5 @@ open_dataset(tf4, partitioning = schema(product_id = 
string()))
 }
 \seealso{
 \href{https://arrow.apache.org/docs/r/articles/dataset.html}{
-datasets article}
+datasets article}, \code{\link[=add_filename]{add_filename()}}
 }
diff --git a/r/tests/testthat/test-dataset.R b/r/tests/testthat/test-dataset.R
index 75d8750e0b7..a1507838b60 100644
--- a/r/tests/testthat/test-dataset.R
+++ b/r/tests/testthat/test-dataset.R
@@ -1504,8 +1504,8 @@ test_that("can add in augmented fields", {
   error_regex <- paste(
     "`add_filename()` or use of the `__filename` augmented field can only",
     "be used with Dataset objects, can only be added before doing",
-    "an aggregation or a join, and cannot be referenced in subsequent",
-    "pipeline steps until either compute() or collect() is called."
+    "an aggregation or a join, and cannot be used in filter() until",
+    "either compute() or collect() is called."
   )
 
   # errors appropriately with ArrowTabular objects
diff --git a/r/vignettes/dataset.Rmd b/r/vignettes/dataset.Rmd
index 36e75963f89..45a42b313d2 100644
--- a/r/vignettes/dataset.Rmd
+++ b/r/vignettes/dataset.Rmd
@@ -308,6 +308,23 @@ in order to declare the types of the virtual columns that 
define the partitions.
 This would be useful, in the NYC taxi data example, if you wanted to keep
 `month` as a string instead of an integer.
 
+### Record which file each row came from
+
+Partitioning only recovers information encoded in directory names. If you need
+the full path of the file each row was read from, use `add_filename()` inside a
+query on the Dataset:
+
+```r
+ds |>
+  mutate(file = add_filename()) |>
+  collect()
+```
+
+This is handy when you have opened a single partition directory directly (for
+example, `open_dataset("nyc-taxi/year=2015")`), since in that case the 
partition
+columns above that directory (here, `year`) are not inferred and you may want 
to
+recover them from the path. See `add_filename()` for details and limitations.
+
 ### Work with multiple data sources
 
 Another feature of Datasets is that they can be composed of multiple data 
sources.

Reply via email to