This is an automated email from the ASF dual-hosted git repository.

thisisnic pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow.git


The following commit(s) were added to refs/heads/main by this push:
     new 9479e686b9f GH-39660: [R] Document partition value types (#50857)
9479e686b9f is described below

commit 9479e686b9fca57f5ce90db704c5eccf3d5a0d48
Author: tomatotomata <[email protected]>
AuthorDate: Wed Sep 30 14:47:03 2026 +0300

    GH-39660: [R] Document partition value types (#50857)
    
    ## Rationale for this change
    
    Partition values such as `001` can be inferred as integers and lose 
meaningful leading zeros. The guidance belongs with the `open_dataset()` API 
documentation, where users look for partitioning and schema controls, rather 
than in the NYC taxi vignette where the example introduces a separate 
hypothetical dataset.
    
    ## What changes are included in this PR?
    
    - Add the leading-zero partition example to the `open_dataset()` 
Partitioning documentation.
    - Show `hive_partition(product_id = string())` as the explicit schema 
declaration.
    - Remove the duplicated example from the dataset vignette so the vignette 
stays focused on its existing flow.
    
    ## Are these changes tested?
    
    - R parsing of `r/R/dataset.R` passes.
    - `git diff --check` passes.
    - The change keeps the example in roxygen documentation and does not alter 
runtime code. A full R package documentation render was not run in this 
checkout.
    
    ## Are there any user-facing changes?
    
    Yes. `?open_dataset` will explain how to preserve leading zeros in Hive 
partition values, in the API documentation for the argument that controls this 
behavior.
    
    Closes #39660
    * GitHub Issue: #39660
    
    Lead-authored-by: ahmadalguydi <[email protected]>
    Co-authored-by: ahmad <[email protected]>
    Co-authored-by: Nic Crane <[email protected]>
    Signed-off-by: Nic Crane <[email protected]>
---
 r/DESCRIPTION         |   2 +-
 r/NAMESPACE           | 244 ++++++++++++++++++++++++++------------------------
 r/R/dataset.R         |  11 ++-
 r/R/dplyr-funcs-doc.R |   2 +-
 r/man/acero.Rd        |   2 +-
 r/man/open_dataset.Rd |  11 ++-
 6 files changed, 152 insertions(+), 120 deletions(-)

diff --git a/r/DESCRIPTION b/r/DESCRIPTION
index 12494b49cc8..43f489f490e 100644
--- a/r/DESCRIPTION
+++ b/r/DESCRIPTION
@@ -151,4 +151,4 @@ Collate:
     'schema.R'
     'udf.R'
     'util.R'
-Config/roxygen2/version: 8.0.0
+Config/roxygen2/version: 8.1.0
diff --git a/r/NAMESPACE b/r/NAMESPACE
index a96b77fdda8..b4eb093b157 100644
--- a/r/NAMESPACE
+++ b/r/NAMESPACE
@@ -426,122 +426,136 @@ export(write_parquet)
 export(write_to_raw)
 export(write_tsv_dataset)
 importFrom(R6,R6Class)
-importFrom(assertthat,assert_that)
-importFrom(assertthat,is.string)
+importFrom(assertthat,
+  assert_that,
+  is.string
+)
 importFrom(bit64,integer64)
 importFrom(glue,glue)
 importFrom(methods,as)
-importFrom(purrr,as_mapper)
-importFrom(purrr,compact)
-importFrom(purrr,flatten)
-importFrom(purrr,imap)
-importFrom(purrr,imap_chr)
-importFrom(purrr,keep)
-importFrom(purrr,map)
-importFrom(purrr,map2)
-importFrom(purrr,map2_chr)
-importFrom(purrr,map_chr)
-importFrom(purrr,map_dbl)
-importFrom(purrr,map_dfr)
-importFrom(purrr,map_int)
-importFrom(purrr,map_lgl)
-importFrom(purrr,reduce)
-importFrom(purrr,walk)
-importFrom(rlang,"%||%")
-importFrom(rlang,":=")
-importFrom(rlang,.data)
-importFrom(rlang,abort)
-importFrom(rlang,arg_match)
-importFrom(rlang,as_function)
-importFrom(rlang,as_label)
-importFrom(rlang,as_quosure)
-importFrom(rlang,call2)
-importFrom(rlang,call_args)
-importFrom(rlang,call_name)
-importFrom(rlang,caller_env)
-importFrom(rlang,check_dots_empty)
-importFrom(rlang,check_dots_empty0)
-importFrom(rlang,dots_list)
-importFrom(rlang,dots_n)
-importFrom(rlang,enexpr)
-importFrom(rlang,enexprs)
-importFrom(rlang,enquo)
-importFrom(rlang,enquos)
-importFrom(rlang,env)
-importFrom(rlang,env_bind)
-importFrom(rlang,eval_tidy)
-importFrom(rlang,exec)
-importFrom(rlang,expr)
-importFrom(rlang,expr_text)
-importFrom(rlang,f_env)
-importFrom(rlang,f_rhs)
-importFrom(rlang,inform)
-importFrom(rlang,is_bare_character)
-importFrom(rlang,is_bare_list)
-importFrom(rlang,is_call)
-importFrom(rlang,is_character)
-importFrom(rlang,is_empty)
-importFrom(rlang,is_false)
-importFrom(rlang,is_formula)
-importFrom(rlang,is_integerish)
-importFrom(rlang,is_interactive)
-importFrom(rlang,is_list)
-importFrom(rlang,is_quosure)
-importFrom(rlang,is_string)
-importFrom(rlang,is_symbol)
-importFrom(rlang,list2)
-importFrom(rlang,new_data_mask)
-importFrom(rlang,new_environment)
-importFrom(rlang,new_quosure)
-importFrom(rlang,new_quosures)
-importFrom(rlang,parse_expr)
-importFrom(rlang,quo)
-importFrom(rlang,quo_get_env)
-importFrom(rlang,quo_get_expr)
-importFrom(rlang,quo_is_call)
-importFrom(rlang,quo_is_null)
-importFrom(rlang,quo_name)
-importFrom(rlang,quo_set_env)
-importFrom(rlang,quo_set_expr)
-importFrom(rlang,quos)
-importFrom(rlang,seq2)
-importFrom(rlang,set_names)
-importFrom(rlang,sym)
-importFrom(rlang,syms)
-importFrom(rlang,trace_back)
-importFrom(rlang,warn)
-importFrom(stats,median)
-importFrom(stats,na.exclude)
-importFrom(stats,na.fail)
-importFrom(stats,na.omit)
-importFrom(stats,na.pass)
-importFrom(stats,quantile)
-importFrom(stats,runif)
-importFrom(tidyselect,all_of)
-importFrom(tidyselect,contains)
-importFrom(tidyselect,ends_with)
-importFrom(tidyselect,eval_rename)
-importFrom(tidyselect,eval_select)
-importFrom(tidyselect,everything)
-importFrom(tidyselect,last_col)
-importFrom(tidyselect,matches)
-importFrom(tidyselect,num_range)
-importFrom(tidyselect,one_of)
-importFrom(tidyselect,starts_with)
-importFrom(tidyselect,vars_pull)
-importFrom(utils,capture.output)
-importFrom(utils,download.file)
-importFrom(utils,getFromNamespace)
-importFrom(utils,head)
-importFrom(utils,install.packages)
-importFrom(utils,modifyList)
-importFrom(utils,object.size)
-importFrom(utils,packageVersion)
-importFrom(utils,tail)
-importFrom(vctrs,s3_register)
-importFrom(vctrs,vec_cast)
-importFrom(vctrs,vec_ptype_abbr)
-importFrom(vctrs,vec_ptype_full)
-importFrom(vctrs,vec_size)
-importFrom(vctrs,vec_unique)
+importFrom(purrr,
+  as_mapper,
+  compact,
+  flatten,
+  imap,
+  imap_chr,
+  keep,
+  map,
+  map2,
+  map2_chr,
+  map_chr,
+  map_dbl,
+  map_dfr,
+  map_int,
+  map_lgl,
+  reduce,
+  walk
+)
+importFrom(rlang,
+  "%||%",
+  ":=",
+  .data,
+  abort,
+  arg_match,
+  as_function,
+  as_label,
+  as_quosure,
+  call2,
+  call_args,
+  call_name,
+  caller_env,
+  check_dots_empty,
+  check_dots_empty0,
+  dots_list,
+  dots_n,
+  enexpr,
+  enexprs,
+  enquo,
+  enquos,
+  env,
+  env_bind,
+  eval_tidy,
+  exec,
+  expr,
+  expr_text,
+  f_env,
+  f_rhs,
+  inform,
+  is_bare_character,
+  is_bare_list,
+  is_call,
+  is_character,
+  is_empty,
+  is_false,
+  is_formula,
+  is_integerish,
+  is_interactive,
+  is_list,
+  is_quosure,
+  is_string,
+  is_symbol,
+  list2,
+  new_data_mask,
+  new_environment,
+  new_quosure,
+  new_quosures,
+  parse_expr,
+  quo,
+  quo_get_env,
+  quo_get_expr,
+  quo_is_call,
+  quo_is_null,
+  quo_name,
+  quo_set_env,
+  quo_set_expr,
+  quos,
+  seq2,
+  set_names,
+  sym,
+  syms,
+  trace_back,
+  warn
+)
+importFrom(stats,
+  median,
+  na.exclude,
+  na.fail,
+  na.omit,
+  na.pass,
+  quantile,
+  runif
+)
+importFrom(tidyselect,
+  all_of,
+  contains,
+  ends_with,
+  eval_rename,
+  eval_select,
+  everything,
+  last_col,
+  matches,
+  num_range,
+  one_of,
+  starts_with,
+  vars_pull
+)
+importFrom(utils,
+  capture.output,
+  download.file,
+  getFromNamespace,
+  head,
+  install.packages,
+  modifyList,
+  object.size,
+  packageVersion,
+  tail
+)
+importFrom(vctrs,
+  s3_register,
+  vec_cast,
+  vec_ptype_abbr,
+  vec_ptype_full,
+  vec_size,
+  vec_unique
+)
 useDynLib(arrow, .registration = TRUE)
diff --git a/r/R/dataset.R b/r/R/dataset.R
index 4ccf338d267..d58ea7d984d 100644
--- a/r/R/dataset.R
+++ b/r/R/dataset.R
@@ -55,7 +55,9 @@
 #' dataset.). If you provide a `Schema` and the names match what is detected,
 #' it will use the types defined by the Schema. In the example file path above,
 #' you could provide a Schema to specify that "month" should be `int8()`
-#' instead of the `int32()` it will be parsed as by default.
+#' instead of the `int32()` it will be parsed as by default. This is also
+#' useful for keeping leading zeros, so that a value such as `001` isn't
+#' parsed as the integer `1`.
 #'
 #' If your file paths do not appear to be Hive-style, or if you pass
 #' `hive_style = FALSE`, the `partitioning` argument will be used to create
@@ -171,6 +173,13 @@
 #'
 #' # If you want to specify the data types for your fields, you can pass in a 
Schema
 #' open_dataset(tf3, partitioning = schema(Month = int8(), Day = int8()))
+#'
+#' # Specifying the type also keeps leading zeros, so "001" stays a string
+#' # instead of becoming the integer 1
+#' products <- data.frame(x = 1:3, product_id = c("001", "002", "010"))
+#' tf4 <- tempfile()
+#' write_dataset(products, tf4, partitioning = "product_id")
+#' open_dataset(tf4, partitioning = schema(product_id = string()))
 open_dataset <- function(
   sources,
   schema = NULL,
diff --git a/r/R/dplyr-funcs-doc.R b/r/R/dplyr-funcs-doc.R
index 1adf23fba1b..61dbf618d32 100644
--- a/r/R/dplyr-funcs-doc.R
+++ b/r/R/dplyr-funcs-doc.R
@@ -84,7 +84,7 @@
 #' Functions can be called either as `pkg::fun()` or just `fun()`, i.e. both
 #' `str_sub()` and `stringr::str_sub()` work.
 #'
-#' In addition to these functions, you can call any of Arrow's 281 compute
+#' In addition to these functions, you can call any of Arrow's 283 compute
 #' functions directly. Arrow has many functions that don't map to an existing R
 #' function. In other cases where there is an R function mapping, you can still
 #' call the Arrow function directly if you don't want the adaptations that the 
R
diff --git a/r/man/acero.Rd b/r/man/acero.Rd
index 0cd6e284e44..1203fe4c43a 100644
--- a/r/man/acero.Rd
+++ b/r/man/acero.Rd
@@ -72,7 +72,7 @@ can assume that the function works in Acero just as it does 
in R.
 Functions can be called either as \code{pkg::fun()} or just \code{fun()}, i.e. 
both
 \code{str_sub()} and \code{stringr::str_sub()} work.
 
-In addition to these functions, you can call any of Arrow's 254 compute
+In addition to these functions, you can call any of Arrow's 283 compute
 functions directly. Arrow has many functions that don't map to an existing R
 function. In other cases where there is an R function mapping, you can still
 call the Arrow function directly if you don't want the adaptations that the R
diff --git a/r/man/open_dataset.Rd b/r/man/open_dataset.Rd
index e2707212d03..93ab25ed527 100644
--- a/r/man/open_dataset.Rd
+++ b/r/man/open_dataset.Rd
@@ -151,7 +151,9 @@ partition columns, do that using \code{select()} or 
\code{rename()} after openin
 dataset.). If you provide a \code{Schema} and the names match what is detected,
 it will use the types defined by the Schema. In the example file path above,
 you could provide a Schema to specify that "month" should be \code{int8()}
-instead of the \code{int32()} it will be parsed as by default.
+instead of the \code{int32()} it will be parsed as by default. This is also
+useful for keeping leading zeros, so that a value such as \code{001} isn't
+parsed as the integer \code{1}.
 
 If your file paths do not appear to be Hive-style, or if you pass
 \code{hive_style = FALSE}, the \code{partitioning} argument will be used to 
create
@@ -206,6 +208,13 @@ open_dataset(tf3, partitioning = c("Month", "Day"))
 
 # If you want to specify the data types for your fields, you can pass in a 
Schema
 open_dataset(tf3, partitioning = schema(Month = int8(), Day = int8()))
+
+# Specifying the type also keeps leading zeros, so "001" stays a string
+# instead of becoming the integer 1
+products <- data.frame(x = 1:3, product_id = c("001", "002", "010"))
+tf4 <- tempfile()
+write_dataset(products, tf4, partitioning = "product_id")
+open_dataset(tf4, partitioning = schema(product_id = string()))
 \dontshow{\}) # examplesIf}
 }
 \seealso{

Reply via email to