This is an automated email from the ASF dual-hosted git repository.
thisisnic pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow.git
The following commit(s) were added to refs/heads/main by this push:
new 9479e686b9f GH-39660: [R] Document partition value types (#50857)
9479e686b9f is described below
commit 9479e686b9fca57f5ce90db704c5eccf3d5a0d48
Author: tomatotomata <[email protected]>
AuthorDate: Wed Sep 30 14:47:03 2026 +0300
GH-39660: [R] Document partition value types (#50857)
## Rationale for this change
Partition values such as `001` can be inferred as integers and lose
meaningful leading zeros. The guidance belongs with the `open_dataset()` API
documentation, where users look for partitioning and schema controls, rather
than in the NYC taxi vignette where the example introduces a separate
hypothetical dataset.
## What changes are included in this PR?
- Add the leading-zero partition example to the `open_dataset()`
Partitioning documentation.
- Show `hive_partition(product_id = string())` as the explicit schema
declaration.
- Remove the duplicated example from the dataset vignette so the vignette
stays focused on its existing flow.
## Are these changes tested?
- R parsing of `r/R/dataset.R` passes.
- `git diff --check` passes.
- The change keeps the example in roxygen documentation and does not alter
runtime code. A full R package documentation render was not run in this
checkout.
## Are there any user-facing changes?
Yes. `?open_dataset` will explain how to preserve leading zeros in Hive
partition values, in the API documentation for the argument that controls this
behavior.
Closes #39660
* GitHub Issue: #39660
Lead-authored-by: ahmadalguydi <[email protected]>
Co-authored-by: ahmad <[email protected]>
Co-authored-by: Nic Crane <[email protected]>
Signed-off-by: Nic Crane <[email protected]>
---
r/DESCRIPTION | 2 +-
r/NAMESPACE | 244 ++++++++++++++++++++++++++------------------------
r/R/dataset.R | 11 ++-
r/R/dplyr-funcs-doc.R | 2 +-
r/man/acero.Rd | 2 +-
r/man/open_dataset.Rd | 11 ++-
6 files changed, 152 insertions(+), 120 deletions(-)
diff --git a/r/DESCRIPTION b/r/DESCRIPTION
index 12494b49cc8..43f489f490e 100644
--- a/r/DESCRIPTION
+++ b/r/DESCRIPTION
@@ -151,4 +151,4 @@ Collate:
'schema.R'
'udf.R'
'util.R'
-Config/roxygen2/version: 8.0.0
+Config/roxygen2/version: 8.1.0
diff --git a/r/NAMESPACE b/r/NAMESPACE
index a96b77fdda8..b4eb093b157 100644
--- a/r/NAMESPACE
+++ b/r/NAMESPACE
@@ -426,122 +426,136 @@ export(write_parquet)
export(write_to_raw)
export(write_tsv_dataset)
importFrom(R6,R6Class)
-importFrom(assertthat,assert_that)
-importFrom(assertthat,is.string)
+importFrom(assertthat,
+ assert_that,
+ is.string
+)
importFrom(bit64,integer64)
importFrom(glue,glue)
importFrom(methods,as)
-importFrom(purrr,as_mapper)
-importFrom(purrr,compact)
-importFrom(purrr,flatten)
-importFrom(purrr,imap)
-importFrom(purrr,imap_chr)
-importFrom(purrr,keep)
-importFrom(purrr,map)
-importFrom(purrr,map2)
-importFrom(purrr,map2_chr)
-importFrom(purrr,map_chr)
-importFrom(purrr,map_dbl)
-importFrom(purrr,map_dfr)
-importFrom(purrr,map_int)
-importFrom(purrr,map_lgl)
-importFrom(purrr,reduce)
-importFrom(purrr,walk)
-importFrom(rlang,"%||%")
-importFrom(rlang,":=")
-importFrom(rlang,.data)
-importFrom(rlang,abort)
-importFrom(rlang,arg_match)
-importFrom(rlang,as_function)
-importFrom(rlang,as_label)
-importFrom(rlang,as_quosure)
-importFrom(rlang,call2)
-importFrom(rlang,call_args)
-importFrom(rlang,call_name)
-importFrom(rlang,caller_env)
-importFrom(rlang,check_dots_empty)
-importFrom(rlang,check_dots_empty0)
-importFrom(rlang,dots_list)
-importFrom(rlang,dots_n)
-importFrom(rlang,enexpr)
-importFrom(rlang,enexprs)
-importFrom(rlang,enquo)
-importFrom(rlang,enquos)
-importFrom(rlang,env)
-importFrom(rlang,env_bind)
-importFrom(rlang,eval_tidy)
-importFrom(rlang,exec)
-importFrom(rlang,expr)
-importFrom(rlang,expr_text)
-importFrom(rlang,f_env)
-importFrom(rlang,f_rhs)
-importFrom(rlang,inform)
-importFrom(rlang,is_bare_character)
-importFrom(rlang,is_bare_list)
-importFrom(rlang,is_call)
-importFrom(rlang,is_character)
-importFrom(rlang,is_empty)
-importFrom(rlang,is_false)
-importFrom(rlang,is_formula)
-importFrom(rlang,is_integerish)
-importFrom(rlang,is_interactive)
-importFrom(rlang,is_list)
-importFrom(rlang,is_quosure)
-importFrom(rlang,is_string)
-importFrom(rlang,is_symbol)
-importFrom(rlang,list2)
-importFrom(rlang,new_data_mask)
-importFrom(rlang,new_environment)
-importFrom(rlang,new_quosure)
-importFrom(rlang,new_quosures)
-importFrom(rlang,parse_expr)
-importFrom(rlang,quo)
-importFrom(rlang,quo_get_env)
-importFrom(rlang,quo_get_expr)
-importFrom(rlang,quo_is_call)
-importFrom(rlang,quo_is_null)
-importFrom(rlang,quo_name)
-importFrom(rlang,quo_set_env)
-importFrom(rlang,quo_set_expr)
-importFrom(rlang,quos)
-importFrom(rlang,seq2)
-importFrom(rlang,set_names)
-importFrom(rlang,sym)
-importFrom(rlang,syms)
-importFrom(rlang,trace_back)
-importFrom(rlang,warn)
-importFrom(stats,median)
-importFrom(stats,na.exclude)
-importFrom(stats,na.fail)
-importFrom(stats,na.omit)
-importFrom(stats,na.pass)
-importFrom(stats,quantile)
-importFrom(stats,runif)
-importFrom(tidyselect,all_of)
-importFrom(tidyselect,contains)
-importFrom(tidyselect,ends_with)
-importFrom(tidyselect,eval_rename)
-importFrom(tidyselect,eval_select)
-importFrom(tidyselect,everything)
-importFrom(tidyselect,last_col)
-importFrom(tidyselect,matches)
-importFrom(tidyselect,num_range)
-importFrom(tidyselect,one_of)
-importFrom(tidyselect,starts_with)
-importFrom(tidyselect,vars_pull)
-importFrom(utils,capture.output)
-importFrom(utils,download.file)
-importFrom(utils,getFromNamespace)
-importFrom(utils,head)
-importFrom(utils,install.packages)
-importFrom(utils,modifyList)
-importFrom(utils,object.size)
-importFrom(utils,packageVersion)
-importFrom(utils,tail)
-importFrom(vctrs,s3_register)
-importFrom(vctrs,vec_cast)
-importFrom(vctrs,vec_ptype_abbr)
-importFrom(vctrs,vec_ptype_full)
-importFrom(vctrs,vec_size)
-importFrom(vctrs,vec_unique)
+importFrom(purrr,
+ as_mapper,
+ compact,
+ flatten,
+ imap,
+ imap_chr,
+ keep,
+ map,
+ map2,
+ map2_chr,
+ map_chr,
+ map_dbl,
+ map_dfr,
+ map_int,
+ map_lgl,
+ reduce,
+ walk
+)
+importFrom(rlang,
+ "%||%",
+ ":=",
+ .data,
+ abort,
+ arg_match,
+ as_function,
+ as_label,
+ as_quosure,
+ call2,
+ call_args,
+ call_name,
+ caller_env,
+ check_dots_empty,
+ check_dots_empty0,
+ dots_list,
+ dots_n,
+ enexpr,
+ enexprs,
+ enquo,
+ enquos,
+ env,
+ env_bind,
+ eval_tidy,
+ exec,
+ expr,
+ expr_text,
+ f_env,
+ f_rhs,
+ inform,
+ is_bare_character,
+ is_bare_list,
+ is_call,
+ is_character,
+ is_empty,
+ is_false,
+ is_formula,
+ is_integerish,
+ is_interactive,
+ is_list,
+ is_quosure,
+ is_string,
+ is_symbol,
+ list2,
+ new_data_mask,
+ new_environment,
+ new_quosure,
+ new_quosures,
+ parse_expr,
+ quo,
+ quo_get_env,
+ quo_get_expr,
+ quo_is_call,
+ quo_is_null,
+ quo_name,
+ quo_set_env,
+ quo_set_expr,
+ quos,
+ seq2,
+ set_names,
+ sym,
+ syms,
+ trace_back,
+ warn
+)
+importFrom(stats,
+ median,
+ na.exclude,
+ na.fail,
+ na.omit,
+ na.pass,
+ quantile,
+ runif
+)
+importFrom(tidyselect,
+ all_of,
+ contains,
+ ends_with,
+ eval_rename,
+ eval_select,
+ everything,
+ last_col,
+ matches,
+ num_range,
+ one_of,
+ starts_with,
+ vars_pull
+)
+importFrom(utils,
+ capture.output,
+ download.file,
+ getFromNamespace,
+ head,
+ install.packages,
+ modifyList,
+ object.size,
+ packageVersion,
+ tail
+)
+importFrom(vctrs,
+ s3_register,
+ vec_cast,
+ vec_ptype_abbr,
+ vec_ptype_full,
+ vec_size,
+ vec_unique
+)
useDynLib(arrow, .registration = TRUE)
diff --git a/r/R/dataset.R b/r/R/dataset.R
index 4ccf338d267..d58ea7d984d 100644
--- a/r/R/dataset.R
+++ b/r/R/dataset.R
@@ -55,7 +55,9 @@
#' dataset.). If you provide a `Schema` and the names match what is detected,
#' it will use the types defined by the Schema. In the example file path above,
#' you could provide a Schema to specify that "month" should be `int8()`
-#' instead of the `int32()` it will be parsed as by default.
+#' instead of the `int32()` it will be parsed as by default. This is also
+#' useful for keeping leading zeros, so that a value such as `001` isn't
+#' parsed as the integer `1`.
#'
#' If your file paths do not appear to be Hive-style, or if you pass
#' `hive_style = FALSE`, the `partitioning` argument will be used to create
@@ -171,6 +173,13 @@
#'
#' # If you want to specify the data types for your fields, you can pass in a
Schema
#' open_dataset(tf3, partitioning = schema(Month = int8(), Day = int8()))
+#'
+#' # Specifying the type also keeps leading zeros, so "001" stays a string
+#' # instead of becoming the integer 1
+#' products <- data.frame(x = 1:3, product_id = c("001", "002", "010"))
+#' tf4 <- tempfile()
+#' write_dataset(products, tf4, partitioning = "product_id")
+#' open_dataset(tf4, partitioning = schema(product_id = string()))
open_dataset <- function(
sources,
schema = NULL,
diff --git a/r/R/dplyr-funcs-doc.R b/r/R/dplyr-funcs-doc.R
index 1adf23fba1b..61dbf618d32 100644
--- a/r/R/dplyr-funcs-doc.R
+++ b/r/R/dplyr-funcs-doc.R
@@ -84,7 +84,7 @@
#' Functions can be called either as `pkg::fun()` or just `fun()`, i.e. both
#' `str_sub()` and `stringr::str_sub()` work.
#'
-#' In addition to these functions, you can call any of Arrow's 281 compute
+#' In addition to these functions, you can call any of Arrow's 283 compute
#' functions directly. Arrow has many functions that don't map to an existing R
#' function. In other cases where there is an R function mapping, you can still
#' call the Arrow function directly if you don't want the adaptations that the
R
diff --git a/r/man/acero.Rd b/r/man/acero.Rd
index 0cd6e284e44..1203fe4c43a 100644
--- a/r/man/acero.Rd
+++ b/r/man/acero.Rd
@@ -72,7 +72,7 @@ can assume that the function works in Acero just as it does
in R.
Functions can be called either as \code{pkg::fun()} or just \code{fun()}, i.e.
both
\code{str_sub()} and \code{stringr::str_sub()} work.
-In addition to these functions, you can call any of Arrow's 254 compute
+In addition to these functions, you can call any of Arrow's 283 compute
functions directly. Arrow has many functions that don't map to an existing R
function. In other cases where there is an R function mapping, you can still
call the Arrow function directly if you don't want the adaptations that the R
diff --git a/r/man/open_dataset.Rd b/r/man/open_dataset.Rd
index e2707212d03..93ab25ed527 100644
--- a/r/man/open_dataset.Rd
+++ b/r/man/open_dataset.Rd
@@ -151,7 +151,9 @@ partition columns, do that using \code{select()} or
\code{rename()} after openin
dataset.). If you provide a \code{Schema} and the names match what is detected,
it will use the types defined by the Schema. In the example file path above,
you could provide a Schema to specify that "month" should be \code{int8()}
-instead of the \code{int32()} it will be parsed as by default.
+instead of the \code{int32()} it will be parsed as by default. This is also
+useful for keeping leading zeros, so that a value such as \code{001} isn't
+parsed as the integer \code{1}.
If your file paths do not appear to be Hive-style, or if you pass
\code{hive_style = FALSE}, the \code{partitioning} argument will be used to
create
@@ -206,6 +208,13 @@ open_dataset(tf3, partitioning = c("Month", "Day"))
# If you want to specify the data types for your fields, you can pass in a
Schema
open_dataset(tf3, partitioning = schema(Month = int8(), Day = int8()))
+
+# Specifying the type also keeps leading zeros, so "001" stays a string
+# instead of becoming the integer 1
+products <- data.frame(x = 1:3, product_id = c("001", "002", "010"))
+tf4 <- tempfile()
+write_dataset(products, tf4, partitioning = "product_id")
+open_dataset(tf4, partitioning = schema(product_id = string()))
\dontshow{\}) # examplesIf}
}
\seealso{