From 129fe3754bc7a80c9ae4d48096d9fbab44db21a2 Mon Sep 17 00:00:00 2001 From: ahmad Date: Thu, 13 Aug 2026 08:08:20 +0300 Subject: [PATCH 1/4] Document preserving leading zeros in partitions Signed-off-by: ahmad --- r/vignettes/dataset.Rmd | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/r/vignettes/dataset.Rmd b/r/vignettes/dataset.Rmd index 36e75963f89..71af5ed5ede 100644 --- a/r/vignettes/dataset.Rmd +++ b/r/vignettes/dataset.Rmd @@ -104,6 +104,17 @@ ds <- open_dataset("nyc-taxi", partitioning = c("year", "month")) Either way, when you look at the Dataset, you can see that in addition to the columns present in every file, there are also columns `year` and `month`. These columns are not present in the files themselves: they are inferred from the partitioning structure. +If a partition value contains meaningful leading zeros, specify its type explicitly. Otherwise, type inference may interpret a value such as `001` as the integer `1`. For example, use a string field when opening a Hive-partitioned dataset whose `product_id` values include leading zeros: + +```r +ds <- open_dataset( + "products", + partitioning = hive_partition(product_id = string()) +) +``` + +With this schema, a path such as `product_id=001/part-0.parquet` keeps `product_id` as the string `"001"` instead of converting it to `1`. + ```{r, eval = file.exists("nyc-taxi")} ds ``` From 1ea21e05ca4b1f01f994f31c7b816cbfc4b4fc6f Mon Sep 17 00:00:00 2001 From: ahmadalguydi Date: Thu, 10 Sep 2026 20:59:43 +0300 Subject: [PATCH 2/4] GH-39660: [R] Document partition value types Signed-off-by: ahmadalguydi --- r/R/dataset.R | 15 +++++++++++++++ r/vignettes/dataset.Rmd | 11 ----------- 2 files changed, 15 insertions(+), 11 deletions(-) diff --git a/r/R/dataset.R b/r/R/dataset.R index 4ccf338d267..112e81e0d81 100644 --- a/r/R/dataset.R +++ b/r/R/dataset.R @@ -57,6 +57,21 @@ #' you could provide a Schema to specify that "month" should be `int8()` #' instead of the `int32()` it will be parsed as by default. #' +#' If a partition value contains meaningful leading zeros, specify its type +#' explicitly. Otherwise, type inference may interpret a value such as `001` +#' as the integer `1`. For example, use a string field when opening a +#' Hive-partitioned dataset whose `product_id` values include leading zeros: +#' +#' ```r +#' ds <- open_dataset( +#' "products", +#' partitioning = hive_partition(product_id = string()) +#' ) +#' ``` +#' +#' With this schema, a path such as `product_id=001/part-0.parquet` keeps +#' `product_id` as the string `"001"` instead of converting it to `1`. +#' #' If your file paths do not appear to be Hive-style, or if you pass #' `hive_style = FALSE`, the `partitioning` argument will be used to create #' Directory partitioning. A character vector of names is required to create diff --git a/r/vignettes/dataset.Rmd b/r/vignettes/dataset.Rmd index 71af5ed5ede..36e75963f89 100644 --- a/r/vignettes/dataset.Rmd +++ b/r/vignettes/dataset.Rmd @@ -104,17 +104,6 @@ ds <- open_dataset("nyc-taxi", partitioning = c("year", "month")) Either way, when you look at the Dataset, you can see that in addition to the columns present in every file, there are also columns `year` and `month`. These columns are not present in the files themselves: they are inferred from the partitioning structure. -If a partition value contains meaningful leading zeros, specify its type explicitly. Otherwise, type inference may interpret a value such as `001` as the integer `1`. For example, use a string field when opening a Hive-partitioned dataset whose `product_id` values include leading zeros: - -```r -ds <- open_dataset( - "products", - partitioning = hive_partition(product_id = string()) -) -``` - -With this schema, a path such as `product_id=001/part-0.parquet` keeps `product_id` as the string `"001"` instead of converting it to `1`. - ```{r, eval = file.exists("nyc-taxi")} ds ``` From 04ba3e975ae12b8311444522a32e051792c0d2c0 Mon Sep 17 00:00:00 2001 From: ahmadalguydi Date: Thu, 10 Sep 2026 21:52:50 +0300 Subject: [PATCH 3/4] GH-39660: refine partitioning example --- r/R/dataset.R | 24 ++++++++++++------------ r/man/open_dataset.Rd | 15 +++++++++++++++ 2 files changed, 27 insertions(+), 12 deletions(-) diff --git a/r/R/dataset.R b/r/R/dataset.R index 112e81e0d81..4ebdaee6cfb 100644 --- a/r/R/dataset.R +++ b/r/R/dataset.R @@ -59,18 +59,8 @@ #' #' If a partition value contains meaningful leading zeros, specify its type #' explicitly. Otherwise, type inference may interpret a value such as `001` -#' as the integer `1`. For example, use a string field when opening a -#' Hive-partitioned dataset whose `product_id` values include leading zeros: -#' -#' ```r -#' ds <- open_dataset( -#' "products", -#' partitioning = hive_partition(product_id = string()) -#' ) -#' ``` -#' -#' With this schema, a path such as `product_id=001/part-0.parquet` keeps -#' `product_id` as the string `"001"` instead of converting it to `1`. +#' as the integer `1`. See the examples below for a runnable Hive-partitioned +#' dataset whose `product_id` values include leading zeros. #' #' If your file paths do not appear to be Hive-style, or if you pass #' `hive_style = FALSE`, the `partitioning` argument will be used to create @@ -186,6 +176,16 @@ #' #' # If you want to specify the data types for your fields, you can pass in a Schema #' open_dataset(tf3, partitioning = schema(Month = int8(), Day = int8())) +#' +#' # If a partition value contains meaningful leading zeros, specify its type +#' # explicitly so values such as "001" remain strings instead of becoming 1. +#' products <- data.frame(x = 1:3, product_id = c("001", "002", "010")) +#' tf4 <- tempfile() +#' write_dataset(products, tf4, partitioning = "product_id") +#' ds <- open_dataset( +#' tf4, +#' partitioning = hive_partition(product_id = string()) +#' ) open_dataset <- function( sources, schema = NULL, diff --git a/r/man/open_dataset.Rd b/r/man/open_dataset.Rd index e2707212d03..73224a22c61 100644 --- a/r/man/open_dataset.Rd +++ b/r/man/open_dataset.Rd @@ -153,6 +153,11 @@ it will use the types defined by the Schema. In the example file path above, you could provide a Schema to specify that "month" should be \code{int8()} instead of the \code{int32()} it will be parsed as by default. +If a partition value contains meaningful leading zeros, specify its type +explicitly. Otherwise, type inference may interpret a value such as \code{001} +as the integer \code{1}. See the examples below for a runnable Hive-partitioned +dataset whose \code{product_id} values include leading zeros. + If your file paths do not appear to be Hive-style, or if you pass \code{hive_style = FALSE}, the \code{partitioning} argument will be used to create Directory partitioning. A character vector of names is required to create @@ -206,6 +211,16 @@ open_dataset(tf3, partitioning = c("Month", "Day")) # If you want to specify the data types for your fields, you can pass in a Schema open_dataset(tf3, partitioning = schema(Month = int8(), Day = int8())) + +# If a partition value contains meaningful leading zeros, specify its type +# explicitly so values such as "001" remain strings instead of becoming 1. +products <- data.frame(x = 1:3, product_id = c("001", "002", "010")) +tf4 <- tempfile() +write_dataset(products, tf4, partitioning = "product_id") +ds <- open_dataset( + tf4, + partitioning = hive_partition(product_id = string()) +) \dontshow{\}) # examplesIf} } \seealso{ From c1dfe356170819238f398c17c5738c6e7242ccad Mon Sep 17 00:00:00 2001 From: Nic Crane Date: Wed, 30 Sep 2026 12:43:54 +0100 Subject: [PATCH 4/4] Tweak and rebuild docs --- r/DESCRIPTION | 2 +- r/NAMESPACE | 244 ++++++++++++++++++++++-------------------- r/R/dataset.R | 18 ++-- r/R/dplyr-funcs-doc.R | 2 +- r/man/acero.Rd | 2 +- r/man/open_dataset.Rd | 18 ++-- 6 files changed, 144 insertions(+), 142 deletions(-) diff --git a/r/DESCRIPTION b/r/DESCRIPTION index 12494b49cc8..43f489f490e 100644 --- a/r/DESCRIPTION +++ b/r/DESCRIPTION @@ -151,4 +151,4 @@ Collate: 'schema.R' 'udf.R' 'util.R' -Config/roxygen2/version: 8.0.0 +Config/roxygen2/version: 8.1.0 diff --git a/r/NAMESPACE b/r/NAMESPACE index a96b77fdda8..b4eb093b157 100644 --- a/r/NAMESPACE +++ b/r/NAMESPACE @@ -426,122 +426,136 @@ export(write_parquet) export(write_to_raw) export(write_tsv_dataset) importFrom(R6,R6Class) -importFrom(assertthat,assert_that) -importFrom(assertthat,is.string) +importFrom(assertthat, + assert_that, + is.string +) importFrom(bit64,integer64) importFrom(glue,glue) importFrom(methods,as) -importFrom(purrr,as_mapper) -importFrom(purrr,compact) -importFrom(purrr,flatten) -importFrom(purrr,imap) -importFrom(purrr,imap_chr) -importFrom(purrr,keep) -importFrom(purrr,map) -importFrom(purrr,map2) -importFrom(purrr,map2_chr) -importFrom(purrr,map_chr) -importFrom(purrr,map_dbl) -importFrom(purrr,map_dfr) -importFrom(purrr,map_int) -importFrom(purrr,map_lgl) -importFrom(purrr,reduce) -importFrom(purrr,walk) -importFrom(rlang,"%||%") -importFrom(rlang,":=") -importFrom(rlang,.data) -importFrom(rlang,abort) -importFrom(rlang,arg_match) -importFrom(rlang,as_function) -importFrom(rlang,as_label) -importFrom(rlang,as_quosure) -importFrom(rlang,call2) -importFrom(rlang,call_args) -importFrom(rlang,call_name) -importFrom(rlang,caller_env) -importFrom(rlang,check_dots_empty) -importFrom(rlang,check_dots_empty0) -importFrom(rlang,dots_list) -importFrom(rlang,dots_n) -importFrom(rlang,enexpr) -importFrom(rlang,enexprs) -importFrom(rlang,enquo) -importFrom(rlang,enquos) -importFrom(rlang,env) -importFrom(rlang,env_bind) -importFrom(rlang,eval_tidy) -importFrom(rlang,exec) -importFrom(rlang,expr) -importFrom(rlang,expr_text) -importFrom(rlang,f_env) -importFrom(rlang,f_rhs) -importFrom(rlang,inform) -importFrom(rlang,is_bare_character) -importFrom(rlang,is_bare_list) -importFrom(rlang,is_call) -importFrom(rlang,is_character) -importFrom(rlang,is_empty) -importFrom(rlang,is_false) -importFrom(rlang,is_formula) -importFrom(rlang,is_integerish) -importFrom(rlang,is_interactive) -importFrom(rlang,is_list) -importFrom(rlang,is_quosure) -importFrom(rlang,is_string) -importFrom(rlang,is_symbol) -importFrom(rlang,list2) -importFrom(rlang,new_data_mask) -importFrom(rlang,new_environment) -importFrom(rlang,new_quosure) -importFrom(rlang,new_quosures) -importFrom(rlang,parse_expr) -importFrom(rlang,quo) -importFrom(rlang,quo_get_env) -importFrom(rlang,quo_get_expr) -importFrom(rlang,quo_is_call) -importFrom(rlang,quo_is_null) -importFrom(rlang,quo_name) -importFrom(rlang,quo_set_env) -importFrom(rlang,quo_set_expr) -importFrom(rlang,quos) -importFrom(rlang,seq2) -importFrom(rlang,set_names) -importFrom(rlang,sym) -importFrom(rlang,syms) -importFrom(rlang,trace_back) -importFrom(rlang,warn) -importFrom(stats,median) -importFrom(stats,na.exclude) -importFrom(stats,na.fail) -importFrom(stats,na.omit) -importFrom(stats,na.pass) -importFrom(stats,quantile) -importFrom(stats,runif) -importFrom(tidyselect,all_of) -importFrom(tidyselect,contains) -importFrom(tidyselect,ends_with) -importFrom(tidyselect,eval_rename) -importFrom(tidyselect,eval_select) -importFrom(tidyselect,everything) -importFrom(tidyselect,last_col) -importFrom(tidyselect,matches) -importFrom(tidyselect,num_range) -importFrom(tidyselect,one_of) -importFrom(tidyselect,starts_with) -importFrom(tidyselect,vars_pull) -importFrom(utils,capture.output) -importFrom(utils,download.file) -importFrom(utils,getFromNamespace) -importFrom(utils,head) -importFrom(utils,install.packages) -importFrom(utils,modifyList) -importFrom(utils,object.size) -importFrom(utils,packageVersion) -importFrom(utils,tail) -importFrom(vctrs,s3_register) -importFrom(vctrs,vec_cast) -importFrom(vctrs,vec_ptype_abbr) -importFrom(vctrs,vec_ptype_full) -importFrom(vctrs,vec_size) -importFrom(vctrs,vec_unique) +importFrom(purrr, + as_mapper, + compact, + flatten, + imap, + imap_chr, + keep, + map, + map2, + map2_chr, + map_chr, + map_dbl, + map_dfr, + map_int, + map_lgl, + reduce, + walk +) +importFrom(rlang, + "%||%", + ":=", + .data, + abort, + arg_match, + as_function, + as_label, + as_quosure, + call2, + call_args, + call_name, + caller_env, + check_dots_empty, + check_dots_empty0, + dots_list, + dots_n, + enexpr, + enexprs, + enquo, + enquos, + env, + env_bind, + eval_tidy, + exec, + expr, + expr_text, + f_env, + f_rhs, + inform, + is_bare_character, + is_bare_list, + is_call, + is_character, + is_empty, + is_false, + is_formula, + is_integerish, + is_interactive, + is_list, + is_quosure, + is_string, + is_symbol, + list2, + new_data_mask, + new_environment, + new_quosure, + new_quosures, + parse_expr, + quo, + quo_get_env, + quo_get_expr, + quo_is_call, + quo_is_null, + quo_name, + quo_set_env, + quo_set_expr, + quos, + seq2, + set_names, + sym, + syms, + trace_back, + warn +) +importFrom(stats, + median, + na.exclude, + na.fail, + na.omit, + na.pass, + quantile, + runif +) +importFrom(tidyselect, + all_of, + contains, + ends_with, + eval_rename, + eval_select, + everything, + last_col, + matches, + num_range, + one_of, + starts_with, + vars_pull +) +importFrom(utils, + capture.output, + download.file, + getFromNamespace, + head, + install.packages, + modifyList, + object.size, + packageVersion, + tail +) +importFrom(vctrs, + s3_register, + vec_cast, + vec_ptype_abbr, + vec_ptype_full, + vec_size, + vec_unique +) useDynLib(arrow, .registration = TRUE) diff --git a/r/R/dataset.R b/r/R/dataset.R index 4ebdaee6cfb..d58ea7d984d 100644 --- a/r/R/dataset.R +++ b/r/R/dataset.R @@ -55,12 +55,9 @@ #' dataset.). If you provide a `Schema` and the names match what is detected, #' it will use the types defined by the Schema. In the example file path above, #' you could provide a Schema to specify that "month" should be `int8()` -#' instead of the `int32()` it will be parsed as by default. -#' -#' If a partition value contains meaningful leading zeros, specify its type -#' explicitly. Otherwise, type inference may interpret a value such as `001` -#' as the integer `1`. See the examples below for a runnable Hive-partitioned -#' dataset whose `product_id` values include leading zeros. +#' instead of the `int32()` it will be parsed as by default. This is also +#' useful for keeping leading zeros, so that a value such as `001` isn't +#' parsed as the integer `1`. #' #' If your file paths do not appear to be Hive-style, or if you pass #' `hive_style = FALSE`, the `partitioning` argument will be used to create @@ -177,15 +174,12 @@ #' # If you want to specify the data types for your fields, you can pass in a Schema #' open_dataset(tf3, partitioning = schema(Month = int8(), Day = int8())) #' -#' # If a partition value contains meaningful leading zeros, specify its type -#' # explicitly so values such as "001" remain strings instead of becoming 1. +#' # Specifying the type also keeps leading zeros, so "001" stays a string +#' # instead of becoming the integer 1 #' products <- data.frame(x = 1:3, product_id = c("001", "002", "010")) #' tf4 <- tempfile() #' write_dataset(products, tf4, partitioning = "product_id") -#' ds <- open_dataset( -#' tf4, -#' partitioning = hive_partition(product_id = string()) -#' ) +#' open_dataset(tf4, partitioning = schema(product_id = string())) open_dataset <- function( sources, schema = NULL, diff --git a/r/R/dplyr-funcs-doc.R b/r/R/dplyr-funcs-doc.R index 1adf23fba1b..61dbf618d32 100644 --- a/r/R/dplyr-funcs-doc.R +++ b/r/R/dplyr-funcs-doc.R @@ -84,7 +84,7 @@ #' Functions can be called either as `pkg::fun()` or just `fun()`, i.e. both #' `str_sub()` and `stringr::str_sub()` work. #' -#' In addition to these functions, you can call any of Arrow's 281 compute +#' In addition to these functions, you can call any of Arrow's 283 compute #' functions directly. Arrow has many functions that don't map to an existing R #' function. In other cases where there is an R function mapping, you can still #' call the Arrow function directly if you don't want the adaptations that the R diff --git a/r/man/acero.Rd b/r/man/acero.Rd index 0cd6e284e44..1203fe4c43a 100644 --- a/r/man/acero.Rd +++ b/r/man/acero.Rd @@ -72,7 +72,7 @@ can assume that the function works in Acero just as it does in R. Functions can be called either as \code{pkg::fun()} or just \code{fun()}, i.e. both \code{str_sub()} and \code{stringr::str_sub()} work. -In addition to these functions, you can call any of Arrow's 254 compute +In addition to these functions, you can call any of Arrow's 283 compute functions directly. Arrow has many functions that don't map to an existing R function. In other cases where there is an R function mapping, you can still call the Arrow function directly if you don't want the adaptations that the R diff --git a/r/man/open_dataset.Rd b/r/man/open_dataset.Rd index 73224a22c61..93ab25ed527 100644 --- a/r/man/open_dataset.Rd +++ b/r/man/open_dataset.Rd @@ -151,12 +151,9 @@ partition columns, do that using \code{select()} or \code{rename()} after openin dataset.). If you provide a \code{Schema} and the names match what is detected, it will use the types defined by the Schema. In the example file path above, you could provide a Schema to specify that "month" should be \code{int8()} -instead of the \code{int32()} it will be parsed as by default. - -If a partition value contains meaningful leading zeros, specify its type -explicitly. Otherwise, type inference may interpret a value such as \code{001} -as the integer \code{1}. See the examples below for a runnable Hive-partitioned -dataset whose \code{product_id} values include leading zeros. +instead of the \code{int32()} it will be parsed as by default. This is also +useful for keeping leading zeros, so that a value such as \code{001} isn't +parsed as the integer \code{1}. If your file paths do not appear to be Hive-style, or if you pass \code{hive_style = FALSE}, the \code{partitioning} argument will be used to create @@ -212,15 +209,12 @@ open_dataset(tf3, partitioning = c("Month", "Day")) # If you want to specify the data types for your fields, you can pass in a Schema open_dataset(tf3, partitioning = schema(Month = int8(), Day = int8())) -# If a partition value contains meaningful leading zeros, specify its type -# explicitly so values such as "001" remain strings instead of becoming 1. +# Specifying the type also keeps leading zeros, so "001" stays a string +# instead of becoming the integer 1 products <- data.frame(x = 1:3, product_id = c("001", "002", "010")) tf4 <- tempfile() write_dataset(products, tf4, partitioning = "product_id") -ds <- open_dataset( - tf4, - partitioning = hive_partition(product_id = string()) -) +open_dataset(tf4, partitioning = schema(product_id = string())) \dontshow{\}) # examplesIf} } \seealso{