Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions r/NEWS.md
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,9 @@
unnested. Similarly, `int64` and `uint32` values inside list columns are
converted to a single R type across the column (#50514).

- `csv_parse_options()` and `CsvParseOptions$create()` gain `pad_short_rows` and
`ignore_extra_columns`, exposing the new C++ parse options (#51659).

# arrow 25.0.1

## Minor improvements and fixes
Expand Down
12 changes: 10 additions & 2 deletions r/R/csv.R
Original file line number Diff line number Diff line change
Expand Up @@ -722,6 +722,10 @@ readr_to_csv_read_options <- function(skip = 0, col_names = TRUE) {
#' and LF (`0x0a`) characters?
#' @param ignore_empty_lines Logical: should empty lines be ignored (default) or
#' generate a row of missing values (if `FALSE`)?
#' @param pad_short_rows Logical: should rows with fewer columns than expected be
#' padded with missing values (default `FALSE`) or raise an error?
#' @param ignore_extra_columns Logical: should columns beyond the number expected be
#' ignored (default `FALSE`) or raise an error?
#' @examplesIf arrow_with_dataset()
#' tf <- tempfile()
#' on.exit(unlink(tf))
Expand All @@ -737,7 +741,9 @@ csv_parse_options <- function(
escaping = FALSE,
escape_char = "\\",
newlines_in_values = FALSE,
ignore_empty_lines = TRUE
ignore_empty_lines = TRUE,
pad_short_rows = FALSE,
ignore_extra_columns = FALSE
) {
csv___ParseOptions__initialize(
list(
Expand All @@ -748,7 +754,9 @@ csv_parse_options <- function(
escaping = escaping,
escape_char = escape_char,
newlines_in_values = newlines_in_values,
ignore_empty_lines = ignore_empty_lines
ignore_empty_lines = ignore_empty_lines,
pad_short_rows = pad_short_rows,
ignore_extra_columns = ignore_extra_columns
)
)
}
Expand Down
2 changes: 2 additions & 0 deletions r/src/csv.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -68,6 +68,8 @@ std::shared_ptr<arrow::csv::ParseOptions> csv___ParseOptions__initialize(
res->escape_char = cpp11::as_cpp<char>(options["escape_char"]);
res->newlines_in_values = cpp11::as_cpp<bool>(options["newlines_in_values"]);
res->ignore_empty_lines = cpp11::as_cpp<bool>(options["ignore_empty_lines"]);
res->pad_short_rows = cpp11::as_cpp<bool>(options["pad_short_rows"]);
res->ignore_extra_columns = cpp11::as_cpp<bool>(options["ignore_extra_columns"]);
return res;
}

Expand Down
48 changes: 48 additions & 0 deletions r/tests/testthat/test-csv.R
Original file line number Diff line number Diff line change
Expand Up @@ -691,6 +691,54 @@ test_that("CSV reading/parsing/convert options can be passed in as lists", {
expect_equal(tab1, tab2)
})

test_that("pad_short_rows and ignore_extra_columns parse options", {
# Rows with fewer columns than expected are padded with nulls (GH-50925), and
# extra columns are dropped (GH-50967).
short <- I("a,b,c\n1,2,9\n3")
expect_error(read_csv_arrow(short))
expect_identical(
read_csv_arrow(short, parse_options = csv_parse_options(pad_short_rows = TRUE)),
tibble::tibble(a = c(1L, 3L), b = c(2L, NA), c = c(9L, NA))
)

extra <- I("a,b\n1,2,3\n4,5")
expect_error(read_csv_arrow(extra))
expect_identical(
read_csv_arrow(extra, parse_options = csv_parse_options(ignore_extra_columns = TRUE)),
tibble::tibble(a = c(1L, 4L), b = c(2L, 5L))
)

# Both together
ragged <- I("a,b\n1,2,3\n4")
expect_identical(
read_csv_arrow(
ragged,
parse_options = csv_parse_options(pad_short_rows = TRUE, ignore_extra_columns = TRUE)
),
tibble::tibble(a = c(1L, 4L), b = c(2L, NA))
)

# The options are also accepted as a plain list, and via the R6 constructor.
expected <- tibble::tibble(a = c(1L, 4L), b = c(2L, NA))
expect_identical(
read_csv_arrow(
ragged,
parse_options = list(pad_short_rows = TRUE, ignore_extra_columns = TRUE)
),
expected
)
expect_identical(
read_csv_arrow(
ragged,
parse_options = CsvParseOptions$create(
pad_short_rows = TRUE,
ignore_extra_columns = TRUE
)
),
expected
)
})

test_that("Read literal data directly", {
expected <- tibble::tibble(x = c(1L, 3L), y = c(2L, 4L))

Expand Down
Loading