diff --git a/r/NEWS.md b/r/NEWS.md index 185316beb8e..9753a9d0ec5 100644 --- a/r/NEWS.md +++ b/r/NEWS.md @@ -27,6 +27,9 @@ unnested. Similarly, `int64` and `uint32` values inside list columns are converted to a single R type across the column (#50514). +- `csv_parse_options()` and `CsvParseOptions$create()` gain `pad_short_rows` and + `ignore_extra_columns`, exposing the new C++ parse options (#51659). + # arrow 25.0.1 ## Minor improvements and fixes diff --git a/r/R/csv.R b/r/R/csv.R index 6506322a9f6..e52b96fdcbd 100644 --- a/r/R/csv.R +++ b/r/R/csv.R @@ -722,6 +722,10 @@ readr_to_csv_read_options <- function(skip = 0, col_names = TRUE) { #' and LF (`0x0a`) characters? #' @param ignore_empty_lines Logical: should empty lines be ignored (default) or #' generate a row of missing values (if `FALSE`)? +#' @param pad_short_rows Logical: should rows with fewer columns than expected be +#' padded with missing values (default `FALSE`) or raise an error? +#' @param ignore_extra_columns Logical: should columns beyond the number expected be +#' ignored (default `FALSE`) or raise an error? #' @examplesIf arrow_with_dataset() #' tf <- tempfile() #' on.exit(unlink(tf)) @@ -737,7 +741,9 @@ csv_parse_options <- function( escaping = FALSE, escape_char = "\\", newlines_in_values = FALSE, - ignore_empty_lines = TRUE + ignore_empty_lines = TRUE, + pad_short_rows = FALSE, + ignore_extra_columns = FALSE ) { csv___ParseOptions__initialize( list( @@ -748,7 +754,9 @@ csv_parse_options <- function( escaping = escaping, escape_char = escape_char, newlines_in_values = newlines_in_values, - ignore_empty_lines = ignore_empty_lines + ignore_empty_lines = ignore_empty_lines, + pad_short_rows = pad_short_rows, + ignore_extra_columns = ignore_extra_columns ) ) } diff --git a/r/src/csv.cpp b/r/src/csv.cpp index d253aa878bc..26a7c618e63 100644 --- a/r/src/csv.cpp +++ b/r/src/csv.cpp @@ -68,6 +68,8 @@ std::shared_ptr csv___ParseOptions__initialize( res->escape_char = cpp11::as_cpp(options["escape_char"]); res->newlines_in_values = cpp11::as_cpp(options["newlines_in_values"]); res->ignore_empty_lines = cpp11::as_cpp(options["ignore_empty_lines"]); + res->pad_short_rows = cpp11::as_cpp(options["pad_short_rows"]); + res->ignore_extra_columns = cpp11::as_cpp(options["ignore_extra_columns"]); return res; } diff --git a/r/tests/testthat/test-csv.R b/r/tests/testthat/test-csv.R index e7da8abd5ce..194cf16df06 100644 --- a/r/tests/testthat/test-csv.R +++ b/r/tests/testthat/test-csv.R @@ -691,6 +691,54 @@ test_that("CSV reading/parsing/convert options can be passed in as lists", { expect_equal(tab1, tab2) }) +test_that("pad_short_rows and ignore_extra_columns parse options", { + # Rows with fewer columns than expected are padded with nulls (GH-50925), and + # extra columns are dropped (GH-50967). + short <- I("a,b,c\n1,2,9\n3") + expect_error(read_csv_arrow(short)) + expect_identical( + read_csv_arrow(short, parse_options = csv_parse_options(pad_short_rows = TRUE)), + tibble::tibble(a = c(1L, 3L), b = c(2L, NA), c = c(9L, NA)) + ) + + extra <- I("a,b\n1,2,3\n4,5") + expect_error(read_csv_arrow(extra)) + expect_identical( + read_csv_arrow(extra, parse_options = csv_parse_options(ignore_extra_columns = TRUE)), + tibble::tibble(a = c(1L, 4L), b = c(2L, 5L)) + ) + + # Both together + ragged <- I("a,b\n1,2,3\n4") + expect_identical( + read_csv_arrow( + ragged, + parse_options = csv_parse_options(pad_short_rows = TRUE, ignore_extra_columns = TRUE) + ), + tibble::tibble(a = c(1L, 4L), b = c(2L, NA)) + ) + + # The options are also accepted as a plain list, and via the R6 constructor. + expected <- tibble::tibble(a = c(1L, 4L), b = c(2L, NA)) + expect_identical( + read_csv_arrow( + ragged, + parse_options = list(pad_short_rows = TRUE, ignore_extra_columns = TRUE) + ), + expected + ) + expect_identical( + read_csv_arrow( + ragged, + parse_options = CsvParseOptions$create( + pad_short_rows = TRUE, + ignore_extra_columns = TRUE + ) + ), + expected + ) +}) + test_that("Read literal data directly", { expected <- tibble::tibble(x = c(1L, 3L), y = c(2L, 4L))