diff --git a/r/DESCRIPTION b/r/DESCRIPTION index 12494b49cc8..43f489f490e 100644 --- a/r/DESCRIPTION +++ b/r/DESCRIPTION @@ -151,4 +151,4 @@ Collate: 'schema.R' 'udf.R' 'util.R' -Config/roxygen2/version: 8.0.0 +Config/roxygen2/version: 8.1.0 diff --git a/r/NAMESPACE b/r/NAMESPACE index a96b77fdda8..b4eb093b157 100644 --- a/r/NAMESPACE +++ b/r/NAMESPACE @@ -426,122 +426,136 @@ export(write_parquet) export(write_to_raw) export(write_tsv_dataset) importFrom(R6,R6Class) -importFrom(assertthat,assert_that) -importFrom(assertthat,is.string) +importFrom(assertthat, + assert_that, + is.string +) importFrom(bit64,integer64) importFrom(glue,glue) importFrom(methods,as) -importFrom(purrr,as_mapper) -importFrom(purrr,compact) -importFrom(purrr,flatten) -importFrom(purrr,imap) -importFrom(purrr,imap_chr) -importFrom(purrr,keep) -importFrom(purrr,map) -importFrom(purrr,map2) -importFrom(purrr,map2_chr) -importFrom(purrr,map_chr) -importFrom(purrr,map_dbl) -importFrom(purrr,map_dfr) -importFrom(purrr,map_int) -importFrom(purrr,map_lgl) -importFrom(purrr,reduce) -importFrom(purrr,walk) -importFrom(rlang,"%||%") -importFrom(rlang,":=") -importFrom(rlang,.data) -importFrom(rlang,abort) -importFrom(rlang,arg_match) -importFrom(rlang,as_function) -importFrom(rlang,as_label) -importFrom(rlang,as_quosure) -importFrom(rlang,call2) -importFrom(rlang,call_args) -importFrom(rlang,call_name) -importFrom(rlang,caller_env) -importFrom(rlang,check_dots_empty) -importFrom(rlang,check_dots_empty0) -importFrom(rlang,dots_list) -importFrom(rlang,dots_n) -importFrom(rlang,enexpr) -importFrom(rlang,enexprs) -importFrom(rlang,enquo) -importFrom(rlang,enquos) -importFrom(rlang,env) -importFrom(rlang,env_bind) -importFrom(rlang,eval_tidy) -importFrom(rlang,exec) -importFrom(rlang,expr) -importFrom(rlang,expr_text) -importFrom(rlang,f_env) -importFrom(rlang,f_rhs) -importFrom(rlang,inform) -importFrom(rlang,is_bare_character) -importFrom(rlang,is_bare_list) -importFrom(rlang,is_call) -importFrom(rlang,is_character) -importFrom(rlang,is_empty) -importFrom(rlang,is_false) -importFrom(rlang,is_formula) -importFrom(rlang,is_integerish) -importFrom(rlang,is_interactive) -importFrom(rlang,is_list) -importFrom(rlang,is_quosure) -importFrom(rlang,is_string) -importFrom(rlang,is_symbol) -importFrom(rlang,list2) -importFrom(rlang,new_data_mask) -importFrom(rlang,new_environment) -importFrom(rlang,new_quosure) -importFrom(rlang,new_quosures) -importFrom(rlang,parse_expr) -importFrom(rlang,quo) -importFrom(rlang,quo_get_env) -importFrom(rlang,quo_get_expr) -importFrom(rlang,quo_is_call) -importFrom(rlang,quo_is_null) -importFrom(rlang,quo_name) -importFrom(rlang,quo_set_env) -importFrom(rlang,quo_set_expr) -importFrom(rlang,quos) -importFrom(rlang,seq2) -importFrom(rlang,set_names) -importFrom(rlang,sym) -importFrom(rlang,syms) -importFrom(rlang,trace_back) -importFrom(rlang,warn) -importFrom(stats,median) -importFrom(stats,na.exclude) -importFrom(stats,na.fail) -importFrom(stats,na.omit) -importFrom(stats,na.pass) -importFrom(stats,quantile) -importFrom(stats,runif) -importFrom(tidyselect,all_of) -importFrom(tidyselect,contains) -importFrom(tidyselect,ends_with) -importFrom(tidyselect,eval_rename) -importFrom(tidyselect,eval_select) -importFrom(tidyselect,everything) -importFrom(tidyselect,last_col) -importFrom(tidyselect,matches) -importFrom(tidyselect,num_range) -importFrom(tidyselect,one_of) -importFrom(tidyselect,starts_with) -importFrom(tidyselect,vars_pull) -importFrom(utils,capture.output) -importFrom(utils,download.file) -importFrom(utils,getFromNamespace) -importFrom(utils,head) -importFrom(utils,install.packages) -importFrom(utils,modifyList) -importFrom(utils,object.size) -importFrom(utils,packageVersion) -importFrom(utils,tail) -importFrom(vctrs,s3_register) -importFrom(vctrs,vec_cast) -importFrom(vctrs,vec_ptype_abbr) -importFrom(vctrs,vec_ptype_full) -importFrom(vctrs,vec_size) -importFrom(vctrs,vec_unique) +importFrom(purrr, + as_mapper, + compact, + flatten, + imap, + imap_chr, + keep, + map, + map2, + map2_chr, + map_chr, + map_dbl, + map_dfr, + map_int, + map_lgl, + reduce, + walk +) +importFrom(rlang, + "%||%", + ":=", + .data, + abort, + arg_match, + as_function, + as_label, + as_quosure, + call2, + call_args, + call_name, + caller_env, + check_dots_empty, + check_dots_empty0, + dots_list, + dots_n, + enexpr, + enexprs, + enquo, + enquos, + env, + env_bind, + eval_tidy, + exec, + expr, + expr_text, + f_env, + f_rhs, + inform, + is_bare_character, + is_bare_list, + is_call, + is_character, + is_empty, + is_false, + is_formula, + is_integerish, + is_interactive, + is_list, + is_quosure, + is_string, + is_symbol, + list2, + new_data_mask, + new_environment, + new_quosure, + new_quosures, + parse_expr, + quo, + quo_get_env, + quo_get_expr, + quo_is_call, + quo_is_null, + quo_name, + quo_set_env, + quo_set_expr, + quos, + seq2, + set_names, + sym, + syms, + trace_back, + warn +) +importFrom(stats, + median, + na.exclude, + na.fail, + na.omit, + na.pass, + quantile, + runif +) +importFrom(tidyselect, + all_of, + contains, + ends_with, + eval_rename, + eval_select, + everything, + last_col, + matches, + num_range, + one_of, + starts_with, + vars_pull +) +importFrom(utils, + capture.output, + download.file, + getFromNamespace, + head, + install.packages, + modifyList, + object.size, + packageVersion, + tail +) +importFrom(vctrs, + s3_register, + vec_cast, + vec_ptype_abbr, + vec_ptype_full, + vec_size, + vec_unique +) useDynLib(arrow, .registration = TRUE) diff --git a/r/R/dataset.R b/r/R/dataset.R index 4ccf338d267..d58ea7d984d 100644 --- a/r/R/dataset.R +++ b/r/R/dataset.R @@ -55,7 +55,9 @@ #' dataset.). If you provide a `Schema` and the names match what is detected, #' it will use the types defined by the Schema. In the example file path above, #' you could provide a Schema to specify that "month" should be `int8()` -#' instead of the `int32()` it will be parsed as by default. +#' instead of the `int32()` it will be parsed as by default. This is also +#' useful for keeping leading zeros, so that a value such as `001` isn't +#' parsed as the integer `1`. #' #' If your file paths do not appear to be Hive-style, or if you pass #' `hive_style = FALSE`, the `partitioning` argument will be used to create @@ -171,6 +173,13 @@ #' #' # If you want to specify the data types for your fields, you can pass in a Schema #' open_dataset(tf3, partitioning = schema(Month = int8(), Day = int8())) +#' +#' # Specifying the type also keeps leading zeros, so "001" stays a string +#' # instead of becoming the integer 1 +#' products <- data.frame(x = 1:3, product_id = c("001", "002", "010")) +#' tf4 <- tempfile() +#' write_dataset(products, tf4, partitioning = "product_id") +#' open_dataset(tf4, partitioning = schema(product_id = string())) open_dataset <- function( sources, schema = NULL, diff --git a/r/R/dplyr-funcs-doc.R b/r/R/dplyr-funcs-doc.R index 1adf23fba1b..61dbf618d32 100644 --- a/r/R/dplyr-funcs-doc.R +++ b/r/R/dplyr-funcs-doc.R @@ -84,7 +84,7 @@ #' Functions can be called either as `pkg::fun()` or just `fun()`, i.e. both #' `str_sub()` and `stringr::str_sub()` work. #' -#' In addition to these functions, you can call any of Arrow's 281 compute +#' In addition to these functions, you can call any of Arrow's 283 compute #' functions directly. Arrow has many functions that don't map to an existing R #' function. In other cases where there is an R function mapping, you can still #' call the Arrow function directly if you don't want the adaptations that the R diff --git a/r/man/acero.Rd b/r/man/acero.Rd index 0cd6e284e44..1203fe4c43a 100644 --- a/r/man/acero.Rd +++ b/r/man/acero.Rd @@ -72,7 +72,7 @@ can assume that the function works in Acero just as it does in R. Functions can be called either as \code{pkg::fun()} or just \code{fun()}, i.e. both \code{str_sub()} and \code{stringr::str_sub()} work. -In addition to these functions, you can call any of Arrow's 254 compute +In addition to these functions, you can call any of Arrow's 283 compute functions directly. Arrow has many functions that don't map to an existing R function. In other cases where there is an R function mapping, you can still call the Arrow function directly if you don't want the adaptations that the R diff --git a/r/man/open_dataset.Rd b/r/man/open_dataset.Rd index e2707212d03..93ab25ed527 100644 --- a/r/man/open_dataset.Rd +++ b/r/man/open_dataset.Rd @@ -151,7 +151,9 @@ partition columns, do that using \code{select()} or \code{rename()} after openin dataset.). If you provide a \code{Schema} and the names match what is detected, it will use the types defined by the Schema. In the example file path above, you could provide a Schema to specify that "month" should be \code{int8()} -instead of the \code{int32()} it will be parsed as by default. +instead of the \code{int32()} it will be parsed as by default. This is also +useful for keeping leading zeros, so that a value such as \code{001} isn't +parsed as the integer \code{1}. If your file paths do not appear to be Hive-style, or if you pass \code{hive_style = FALSE}, the \code{partitioning} argument will be used to create @@ -206,6 +208,13 @@ open_dataset(tf3, partitioning = c("Month", "Day")) # If you want to specify the data types for your fields, you can pass in a Schema open_dataset(tf3, partitioning = schema(Month = int8(), Day = int8())) + +# Specifying the type also keeps leading zeros, so "001" stays a string +# instead of becoming the integer 1 +products <- data.frame(x = 1:3, product_id = c("001", "002", "010")) +tf4 <- tempfile() +write_dataset(products, tf4, partitioning = "product_id") +open_dataset(tf4, partitioning = schema(product_id = string())) \dontshow{\}) # examplesIf} } \seealso{