Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
b4df3e7
Add unit tests for write_csv_arrow
thisisnic Apr 22, 2021
1077e8d
Add CsvWriteOptions and write_csv_arrow functions
thisisnic Apr 22, 2021
9d12fe6
Add C++ functions for intialising csv writeoptions and bindings to ca…
thisisnic Apr 22, 2021
5791f63
Add pkgdown reference to CsvWriteOptions
thisisnic Apr 22, 2021
deff496
Add WriteOptions to arrow::csv namespace
thisisnic Apr 23, 2021
e21aaa5
Update types
thisisnic Apr 23, 2021
04d8106
Update arrowExports
thisisnic Apr 23, 2021
59112f6
Re-order params and add assertion
thisisnic Apr 23, 2021
c172db8
try changing
thisisnic Apr 23, 2021
2b962ad
Remove const keyword
thisisnic Apr 23, 2021
1d70f0b
Remove unnecessary comma, include relevant header files, refactor cpp
thisisnic Apr 23, 2021
713075f
Remove exposed memory pool argument, update NAMESPACE and docs
thisisnic Apr 26, 2021
86d83d8
Use gc_memory_pool() instead of arrow::default_memory_pool() to preve…
thisisnic Apr 26, 2021
8de7c3b
Typo fix
thisisnic Apr 26, 2021
3239575
Skip tests that include writing date columns
thisisnic Apr 27, 2021
63a2f98
Change whitespace to force CI
thisisnic Apr 27, 2021
96e9e2e
Change R C++ function format
thisisnic Apr 27, 2021
fcc6f5f
Run linter on R C++
thisisnic Apr 27, 2021
ee38464
Update docs
thisisnic Apr 27, 2021
6b6914c
Add write_csv_arrrow to _pkgdown.yml
thisisnic Apr 28, 2021
93338cc
Inconsequential grammar change to trigger CI
thisisnic Apr 28, 2021
7722206
Remove extra no-dates tests
thisisnic Apr 30, 2021
0d62fa1
Move docs for CsvWriteOptions
thisisnic Apr 30, 2021
faab7ff
Move tests to bottom of file
thisisnic May 3, 2021
8ff781f
Add tests for invalid inputs
thisisnic May 3, 2021
a421243
Remove extra whitespace
thisisnic May 3, 2021
6252dcd
Move batch_size validation
nealrichardson May 4, 2021
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions cpp/src/arrow/csv/type_fwd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -22,6 +22,7 @@ class TableReader;
struct ConvertOptions;
struct ReadOptions;
struct ParseOptions;
struct WriteOptions;

} // namespace csv
} // namespace arrow
2 changes: 2 additions & 0 deletions r/NAMESPACE
Original file line numberDiff line numberDiff line change
Expand Up@@ -122,6 +122,7 @@ export(CsvFragmentScanOptions)
export(CsvParseOptions)
export(CsvReadOptions)
export(CsvTableReader)
export(CsvWriteOptions)
export(Dataset)
export(DatasetFactory)
export(DateUnit)
Expand DownExpand Up@@ -277,6 +278,7 @@ export(unify_schemas)
export(utf8)
export(value_counts)
export(write_arrow)
export(write_csv_arrow)
export(write_dataset)
export(write_feather)
export(write_ipc_stream)
Expand Down
12 changes: 12 additions & 0 deletions r/R/arrowExports.R

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

64 changes: 64 additions & 0 deletions r/R/csv.R
Original file line numberDiff line numberDiff line change
Expand Up@@ -381,6 +381,11 @@ CsvTableReader$create <- function(file,
#' `TimestampParser$create()` takes an optional `format` string argument.
#' See [`strptime()`][base::strptime()] for example syntax.
#' The default is to use an ISO-8601 format parser.
#'
#' The `CsvWriteOptions$create()` factory method takes the following arguments:
#' - `include_header` Whether to write an initial header line with column names
#' - `batch_size` Maximum number of rows processed at a time. Default is 1024.
#'
#' @section Active bindings:
#'
#' - `column_names`: from `CsvReadOptions`
Expand DownExpand Up@@ -408,6 +413,19 @@ CsvReadOptions$create <- function(use_threads = option_use_threads(),
)
}

#' @rdname CsvReadOptions
#' @export
CsvWriteOptions <- R6Class("CsvWriteOptions", inherit = ArrowObject)
CsvWriteOptions$create <- function(include_header = TRUE, batch_size = 1024L){
assert_that(is_integerish(batch_size, n = 1, finite = TRUE), batch_size > 0)
csv___WriteOptions__initialize(
list(
include_header = include_header,
batch_size = as.integer(batch_size)
)
)
}

readr_to_csv_read_options <- function(skip, col_names, col_types) {
if (isTRUE(col_names)) {
# C++ default to parse is 0-length string array
Expand DownExpand Up@@ -585,3 +603,49 @@ readr_to_csv_convert_options <- function(na,
include_columns = include_columns
)
}

#' Write CSV file to disk
#'
#' @param x `data.frame`, [RecordBatch], or [Table]
#' @param sink A string file path, URI, or [OutputStream], or path in a file
#' system (`SubTreeFileSystem`)
#' @param include_header Whether to write an initial header line with column names
#' @param batch_size Maximum number of rows processed at a time. Default is 1024.
#'
#' @return The input `x`, invisibly. Note that if `sink` is an [OutputStream],
#' the stream will be left open.
#' @export
#' @examples
#' \donttest{
#' tf <- tempfile()
#' on.exit(unlink(tf))
#' write_csv_arrow(mtcars, tf)
#' }
#' @include arrow-package.R
write_csv_arrow <- function(x,
sink,
include_header = TRUE,
batch_size = 1024L) {

write_options <- CsvWriteOptions$create(include_header, batch_size)

x_out <- x
if (is.data.frame(x)) {
x <- Table$create(x)
}

assert_is(x, "ArrowTabular")

if (!inherits(sink, "OutputStream")) {
sink <- make_output_stream(sink)
on.exit(sink$close())
}

if(inherits(x, "RecordBatch")){
csv___WriteCSV__RecordBatch(x, write_options, sink)
} else if(inherits(x, "Table")){
csv___WriteCSV__Table(x, write_options, sink)
}

invisible(x_out)
}
2 changes: 2 additions & 0 deletions r/_pkgdown.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -98,6 +98,7 @@ reference:
- write_ipc_stream
- write_to_raw
- write_parquet
- write_csv_arrow
- title: C++ reader/writer interface
contents:
- ParquetFileReader
Expand All@@ -109,6 +110,7 @@ reference:
- RecordBatchReader
- RecordBatchWriter
- CsvReadOptions
- CsvWriteOptions
- title: Arrow data containers
contents:
- array
Expand Down
22 changes: 22 additions & 0 deletions r/man/CsvWriteOptions.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

32 changes: 32 additions & 0 deletions r/man/write_csv_arrow.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

54 changes: 54 additions & 0 deletions r/src/arrowExports.cpp

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions r/src/arrow_types.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -178,6 +178,7 @@ R6_CLASS_NAME(arrow::csv::ReadOptions, "CsvReadOptions");
R6_CLASS_NAME(arrow::csv::ParseOptions, "CsvParseOptions");
R6_CLASS_NAME(arrow::csv::ConvertOptions, "CsvConvertOptions");
R6_CLASS_NAME(arrow::csv::TableReader, "CsvTableReader");
R6_CLASS_NAME(arrow::csv::WriteOptions, "CsvWriteOptions");

#if defined(ARROW_R_WITH_PARQUET)
R6_CLASS_NAME(parquet::ArrowReaderProperties, "ParquetArrowReaderProperties");
Expand Down
30 changes: 30 additions & 0 deletions r/src/csv.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,8 +20,21 @@
#if defined(ARROW_R_WITH_ARROW)

#include <arrow/csv/reader.h>
#include <arrow/csv/writer.h>
#include <arrow/memory_pool.h>

#include <arrow/util/value_parsing.h>

// [[arrow::export]]
std::shared_ptr<arrow::csv::WriteOptions> csv___WriteOptions__initialize(
cpp11::list options) {
auto res =
std::make_shared<arrow::csv::WriteOptions>(arrow::csv::WriteOptions::Defaults());
res->include_header = cpp11::as_cpp<bool>(options["include_header"]);
res->batch_size = cpp11::as_cpp<int>(options["batch_size"]);
return res;
}

// [[arrow::export]]
std::shared_ptr<arrow::csv::ReadOptions> csv___ReadOptions__initialize(
cpp11::list options) {
Expand DownExpand Up@@ -174,4 +187,21 @@ std::shared_ptr<arrow::TimestampParser> TimestampParser__MakeISO8601() {
return arrow::TimestampParser::MakeISO8601();
}

// [[arrow::export]]
void csv___WriteCSV__Table(const std::shared_ptr<arrow::Table>& table,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(
arrow::csv::WriteCSV(*table, *write_options, gc_memory_pool(), stream.get()));
}

// [[arrow::export]]
void csv___WriteCSV__RecordBatch(
const std::shared_ptr<arrow::RecordBatch>& record_batch,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(arrow::csv::WriteCSV(*record_batch, *write_options, gc_memory_pool(),
stream.get()));
}

#endif
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
b4df3e7
Add unit tests for write_csv_arrow
thisisnic Apr 22, 2021
1077e8d
Add CsvWriteOptions and write_csv_arrow functions
thisisnic Apr 22, 2021
9d12fe6
Add C++ functions for intialising csv writeoptions and bindings to ca…
thisisnic Apr 22, 2021
5791f63
Add pkgdown reference to CsvWriteOptions
thisisnic Apr 22, 2021
deff496
Add WriteOptions to arrow::csv namespace
thisisnic Apr 23, 2021
e21aaa5
Update types
thisisnic Apr 23, 2021
04d8106
Update arrowExports
thisisnic Apr 23, 2021
59112f6
Re-order params and add assertion
thisisnic Apr 23, 2021
c172db8
try changing
thisisnic Apr 23, 2021
2b962ad
Remove const keyword
thisisnic Apr 23, 2021
1d70f0b
Remove unnecessary comma, include relevant header files, refactor cpp
thisisnic Apr 23, 2021
713075f
Remove exposed memory pool argument, update NAMESPACE and docs
thisisnic Apr 26, 2021
86d83d8
Use gc_memory_pool() instead of arrow::default_memory_pool() to preve…
thisisnic Apr 26, 2021
8de7c3b
Typo fix
thisisnic Apr 26, 2021
3239575
Skip tests that include writing date columns
thisisnic Apr 27, 2021
63a2f98
Change whitespace to force CI
thisisnic Apr 27, 2021
96e9e2e
Change R C++ function format
thisisnic Apr 27, 2021
fcc6f5f
Run linter on R C++
thisisnic Apr 27, 2021
ee38464
Update docs
thisisnic Apr 27, 2021
6b6914c
Add write_csv_arrrow to _pkgdown.yml
thisisnic Apr 28, 2021
93338cc
Inconsequential grammar change to trigger CI
thisisnic Apr 28, 2021
7722206
Remove extra no-dates tests
thisisnic Apr 30, 2021
0d62fa1
Move docs for CsvWriteOptions
thisisnic Apr 30, 2021
faab7ff
Move tests to bottom of file
thisisnic May 3, 2021
8ff781f
Add tests for invalid inputs
thisisnic May 3, 2021
a421243
Remove extra whitespace
thisisnic May 3, 2021
6252dcd
Move batch_size validation
nealrichardson May 4, 2021
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions cpp/src/arrow/csv/type_fwd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -22,6 +22,7 @@ class TableReader;
struct ConvertOptions;
struct ReadOptions;
struct ParseOptions;
struct WriteOptions;

} // namespace csv
} // namespace arrow
2 changes: 2 additions & 0 deletions r/NAMESPACE
Original file line numberDiff line numberDiff line change
Expand Up@@ -122,6 +122,7 @@ export(CsvFragmentScanOptions)
export(CsvParseOptions)
export(CsvReadOptions)
export(CsvTableReader)
export(CsvWriteOptions)
export(Dataset)
export(DatasetFactory)
export(DateUnit)
Expand DownExpand Up@@ -277,6 +278,7 @@ export(unify_schemas)
export(utf8)
export(value_counts)
export(write_arrow)
export(write_csv_arrow)
export(write_dataset)
export(write_feather)
export(write_ipc_stream)
Expand Down
12 changes: 12 additions & 0 deletions r/R/arrowExports.R

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

64 changes: 64 additions & 0 deletions r/R/csv.R
Original file line numberDiff line numberDiff line change
Expand Up@@ -381,6 +381,11 @@ CsvTableReader$create <- function(file,
#' `TimestampParser$create()` takes an optional `format` string argument.
#' See [`strptime()`][base::strptime()] for example syntax.
#' The default is to use an ISO-8601 format parser.
#'
#' The `CsvWriteOptions$create()` factory method takes the following arguments:
#' - `include_header` Whether to write an initial header line with column names
#' - `batch_size` Maximum number of rows processed at a time. Default is 1024.
#'
#' @section Active bindings:
#'
#' - `column_names`: from `CsvReadOptions`
Expand DownExpand Up@@ -408,6 +413,19 @@ CsvReadOptions$create <- function(use_threads = option_use_threads(),
)
}

#' @rdname CsvReadOptions
#' @export
CsvWriteOptions <- R6Class("CsvWriteOptions", inherit = ArrowObject)
CsvWriteOptions$create <- function(include_header = TRUE, batch_size = 1024L){
assert_that(is_integerish(batch_size, n = 1, finite = TRUE), batch_size > 0)
csv___WriteOptions__initialize(
list(
include_header = include_header,
batch_size = as.integer(batch_size)
)
)
}

readr_to_csv_read_options <- function(skip, col_names, col_types) {
if (isTRUE(col_names)) {
# C++ default to parse is 0-length string array
Expand DownExpand Up@@ -585,3 +603,49 @@ readr_to_csv_convert_options <- function(na,
include_columns = include_columns
)
}

#' Write CSV file to disk
#'
#' @param x `data.frame`, [RecordBatch], or [Table]
#' @param sink A string file path, URI, or [OutputStream], or path in a file
#' system (`SubTreeFileSystem`)
#' @param include_header Whether to write an initial header line with column names
#' @param batch_size Maximum number of rows processed at a time. Default is 1024.
#'
#' @return The input `x`, invisibly. Note that if `sink` is an [OutputStream],
#' the stream will be left open.
#' @export
#' @examples
#' \donttest{
#' tf <- tempfile()
#' on.exit(unlink(tf))
#' write_csv_arrow(mtcars, tf)
#' }
#' @include arrow-package.R
write_csv_arrow <- function(x,
sink,
include_header = TRUE,
batch_size = 1024L) {

write_options <- CsvWriteOptions$create(include_header, batch_size)

x_out <- x
if (is.data.frame(x)) {
x <- Table$create(x)
}

assert_is(x, "ArrowTabular")

if (!inherits(sink, "OutputStream")) {
sink <- make_output_stream(sink)
on.exit(sink$close())
}

if(inherits(x, "RecordBatch")){
csv___WriteCSV__RecordBatch(x, write_options, sink)
} else if(inherits(x, "Table")){
csv___WriteCSV__Table(x, write_options, sink)
}

invisible(x_out)
}
2 changes: 2 additions & 0 deletions r/_pkgdown.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -98,6 +98,7 @@ reference:
- write_ipc_stream
- write_to_raw
- write_parquet
- write_csv_arrow
- title: C++ reader/writer interface
contents:
- ParquetFileReader
Expand All@@ -109,6 +110,7 @@ reference:
- RecordBatchReader
- RecordBatchWriter
- CsvReadOptions
- CsvWriteOptions
- title: Arrow data containers
contents:
- array
Expand Down
22 changes: 22 additions & 0 deletions r/man/CsvWriteOptions.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

32 changes: 32 additions & 0 deletions r/man/write_csv_arrow.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

54 changes: 54 additions & 0 deletions r/src/arrowExports.cpp

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions r/src/arrow_types.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -178,6 +178,7 @@ R6_CLASS_NAME(arrow::csv::ReadOptions, "CsvReadOptions");
R6_CLASS_NAME(arrow::csv::ParseOptions, "CsvParseOptions");
R6_CLASS_NAME(arrow::csv::ConvertOptions, "CsvConvertOptions");
R6_CLASS_NAME(arrow::csv::TableReader, "CsvTableReader");
R6_CLASS_NAME(arrow::csv::WriteOptions, "CsvWriteOptions");

#if defined(ARROW_R_WITH_PARQUET)
R6_CLASS_NAME(parquet::ArrowReaderProperties, "ParquetArrowReaderProperties");
Expand Down
30 changes: 30 additions & 0 deletions r/src/csv.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,8 +20,21 @@
#if defined(ARROW_R_WITH_ARROW)

#include <arrow/csv/reader.h>
#include <arrow/csv/writer.h>
#include <arrow/memory_pool.h>

#include <arrow/util/value_parsing.h>

// [[arrow::export]]
std::shared_ptr<arrow::csv::WriteOptions> csv___WriteOptions__initialize(
cpp11::list options) {
auto res =
std::make_shared<arrow::csv::WriteOptions>(arrow::csv::WriteOptions::Defaults());
res->include_header = cpp11::as_cpp<bool>(options["include_header"]);
res->batch_size = cpp11::as_cpp<int>(options["batch_size"]);
return res;
}

// [[arrow::export]]
std::shared_ptr<arrow::csv::ReadOptions> csv___ReadOptions__initialize(
cpp11::list options) {
Expand DownExpand Up@@ -174,4 +187,21 @@ std::shared_ptr<arrow::TimestampParser> TimestampParser__MakeISO8601() {
return arrow::TimestampParser::MakeISO8601();
}

// [[arrow::export]]
void csv___WriteCSV__Table(const std::shared_ptr<arrow::Table>& table,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(
arrow::csv::WriteCSV(*table, *write_options, gc_memory_pool(), stream.get()));
}

// [[arrow::export]]
void csv___WriteCSV__RecordBatch(
const std::shared_ptr<arrow::RecordBatch>& record_batch,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(arrow::csv::WriteCSV(*record_batch, *write_options, gc_memory_pool(),
stream.get()));
}

#endif
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
b4df3e7
Add unit tests for write_csv_arrow
thisisnic Apr 22, 2021
1077e8d
Add CsvWriteOptions and write_csv_arrow functions
thisisnic Apr 22, 2021
9d12fe6
Add C++ functions for intialising csv writeoptions and bindings to ca…
thisisnic Apr 22, 2021
5791f63
Add pkgdown reference to CsvWriteOptions
thisisnic Apr 22, 2021
deff496
Add WriteOptions to arrow::csv namespace
thisisnic Apr 23, 2021
e21aaa5
Update types
thisisnic Apr 23, 2021
04d8106
Update arrowExports
thisisnic Apr 23, 2021
59112f6
Re-order params and add assertion
thisisnic Apr 23, 2021
c172db8
try changing
thisisnic Apr 23, 2021
2b962ad
Remove const keyword
thisisnic Apr 23, 2021
1d70f0b
Remove unnecessary comma, include relevant header files, refactor cpp
thisisnic Apr 23, 2021
713075f
Remove exposed memory pool argument, update NAMESPACE and docs
thisisnic Apr 26, 2021
86d83d8
Use gc_memory_pool() instead of arrow::default_memory_pool() to preve…
thisisnic Apr 26, 2021
8de7c3b
Typo fix
thisisnic Apr 26, 2021
3239575
Skip tests that include writing date columns
thisisnic Apr 27, 2021
63a2f98
Change whitespace to force CI
thisisnic Apr 27, 2021
96e9e2e
Change R C++ function format
thisisnic Apr 27, 2021
fcc6f5f
Run linter on R C++
thisisnic Apr 27, 2021
ee38464
Update docs
thisisnic Apr 27, 2021
6b6914c
Add write_csv_arrrow to _pkgdown.yml
thisisnic Apr 28, 2021
93338cc
Inconsequential grammar change to trigger CI
thisisnic Apr 28, 2021
7722206
Remove extra no-dates tests
thisisnic Apr 30, 2021
0d62fa1
Move docs for CsvWriteOptions
thisisnic Apr 30, 2021
faab7ff
Move tests to bottom of file
thisisnic May 3, 2021
8ff781f
Add tests for invalid inputs
thisisnic May 3, 2021
a421243
Remove extra whitespace
thisisnic May 3, 2021
6252dcd
Move batch_size validation
nealrichardson May 4, 2021
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions cpp/src/arrow/csv/type_fwd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -22,6 +22,7 @@ class TableReader;
struct ConvertOptions;
struct ReadOptions;
struct ParseOptions;
struct WriteOptions;

} // namespace csv
} // namespace arrow
2 changes: 2 additions & 0 deletions r/NAMESPACE
Original file line numberDiff line numberDiff line change
Expand Up@@ -122,6 +122,7 @@ export(CsvFragmentScanOptions)
export(CsvParseOptions)
export(CsvReadOptions)
export(CsvTableReader)
export(CsvWriteOptions)
export(Dataset)
export(DatasetFactory)
export(DateUnit)
Expand DownExpand Up@@ -277,6 +278,7 @@ export(unify_schemas)
export(utf8)
export(value_counts)
export(write_arrow)
export(write_csv_arrow)
export(write_dataset)
export(write_feather)
export(write_ipc_stream)
Expand Down
12 changes: 12 additions & 0 deletions r/R/arrowExports.R

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

64 changes: 64 additions & 0 deletions r/R/csv.R
Original file line numberDiff line numberDiff line change
Expand Up@@ -381,6 +381,11 @@ CsvTableReader$create <- function(file,
#' `TimestampParser$create()` takes an optional `format` string argument.
#' See [`strptime()`][base::strptime()] for example syntax.
#' The default is to use an ISO-8601 format parser.
#'
#' The `CsvWriteOptions$create()` factory method takes the following arguments:
#' - `include_header` Whether to write an initial header line with column names
#' - `batch_size` Maximum number of rows processed at a time. Default is 1024.
#'
#' @section Active bindings:
#'
#' - `column_names`: from `CsvReadOptions`
Expand DownExpand Up@@ -408,6 +413,19 @@ CsvReadOptions$create <- function(use_threads = option_use_threads(),
)
}

#' @rdname CsvReadOptions
#' @export
CsvWriteOptions <- R6Class("CsvWriteOptions", inherit = ArrowObject)
CsvWriteOptions$create <- function(include_header = TRUE, batch_size = 1024L){
assert_that(is_integerish(batch_size, n = 1, finite = TRUE), batch_size > 0)
csv___WriteOptions__initialize(
list(
include_header = include_header,
batch_size = as.integer(batch_size)
)
)
}

readr_to_csv_read_options <- function(skip, col_names, col_types) {
if (isTRUE(col_names)) {
# C++ default to parse is 0-length string array
Expand DownExpand Up@@ -585,3 +603,49 @@ readr_to_csv_convert_options <- function(na,
include_columns = include_columns
)
}

#' Write CSV file to disk
#'
#' @param x `data.frame`, [RecordBatch], or [Table]
#' @param sink A string file path, URI, or [OutputStream], or path in a file
#' system (`SubTreeFileSystem`)
#' @param include_header Whether to write an initial header line with column names
#' @param batch_size Maximum number of rows processed at a time. Default is 1024.
#'
#' @return The input `x`, invisibly. Note that if `sink` is an [OutputStream],
#' the stream will be left open.
#' @export
#' @examples
#' \donttest{
#' tf <- tempfile()
#' on.exit(unlink(tf))
#' write_csv_arrow(mtcars, tf)
#' }
#' @include arrow-package.R
write_csv_arrow <- function(x,
sink,
include_header = TRUE,
batch_size = 1024L) {

write_options <- CsvWriteOptions$create(include_header, batch_size)

x_out <- x
if (is.data.frame(x)) {
x <- Table$create(x)
}

assert_is(x, "ArrowTabular")

if (!inherits(sink, "OutputStream")) {
sink <- make_output_stream(sink)
on.exit(sink$close())
}

if(inherits(x, "RecordBatch")){
csv___WriteCSV__RecordBatch(x, write_options, sink)
} else if(inherits(x, "Table")){
csv___WriteCSV__Table(x, write_options, sink)
}

invisible(x_out)
}
2 changes: 2 additions & 0 deletions r/_pkgdown.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -98,6 +98,7 @@ reference:
- write_ipc_stream
- write_to_raw
- write_parquet
- write_csv_arrow
- title: C++ reader/writer interface
contents:
- ParquetFileReader
Expand All@@ -109,6 +110,7 @@ reference:
- RecordBatchReader
- RecordBatchWriter
- CsvReadOptions
- CsvWriteOptions
- title: Arrow data containers
contents:
- array
Expand Down
22 changes: 22 additions & 0 deletions r/man/CsvWriteOptions.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

32 changes: 32 additions & 0 deletions r/man/write_csv_arrow.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

54 changes: 54 additions & 0 deletions r/src/arrowExports.cpp

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions r/src/arrow_types.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -178,6 +178,7 @@ R6_CLASS_NAME(arrow::csv::ReadOptions, "CsvReadOptions");
R6_CLASS_NAME(arrow::csv::ParseOptions, "CsvParseOptions");
R6_CLASS_NAME(arrow::csv::ConvertOptions, "CsvConvertOptions");
R6_CLASS_NAME(arrow::csv::TableReader, "CsvTableReader");
R6_CLASS_NAME(arrow::csv::WriteOptions, "CsvWriteOptions");

#if defined(ARROW_R_WITH_PARQUET)
R6_CLASS_NAME(parquet::ArrowReaderProperties, "ParquetArrowReaderProperties");
Expand Down
30 changes: 30 additions & 0 deletions r/src/csv.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,8 +20,21 @@
#if defined(ARROW_R_WITH_ARROW)

#include <arrow/csv/reader.h>
#include <arrow/csv/writer.h>
#include <arrow/memory_pool.h>

#include <arrow/util/value_parsing.h>

// [[arrow::export]]
std::shared_ptr<arrow::csv::WriteOptions> csv___WriteOptions__initialize(
cpp11::list options) {
auto res =
std::make_shared<arrow::csv::WriteOptions>(arrow::csv::WriteOptions::Defaults());
res->include_header = cpp11::as_cpp<bool>(options["include_header"]);
res->batch_size = cpp11::as_cpp<int>(options["batch_size"]);
return res;
}

// [[arrow::export]]
std::shared_ptr<arrow::csv::ReadOptions> csv___ReadOptions__initialize(
cpp11::list options) {
Expand DownExpand Up@@ -174,4 +187,21 @@ std::shared_ptr<arrow::TimestampParser> TimestampParser__MakeISO8601() {
return arrow::TimestampParser::MakeISO8601();
}

// [[arrow::export]]
void csv___WriteCSV__Table(const std::shared_ptr<arrow::Table>& table,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(
arrow::csv::WriteCSV(*table, *write_options, gc_memory_pool(), stream.get()));
}

// [[arrow::export]]
void csv___WriteCSV__RecordBatch(
const std::shared_ptr<arrow::RecordBatch>& record_batch,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(arrow::csv::WriteCSV(*record_batch, *write_options, gc_memory_pool(),
stream.get()));
}

#endif
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
b4df3e7
Add unit tests for write_csv_arrow
thisisnic Apr 22, 2021
1077e8d
Add CsvWriteOptions and write_csv_arrow functions
thisisnic Apr 22, 2021
9d12fe6
Add C++ functions for intialising csv writeoptions and bindings to ca…
thisisnic Apr 22, 2021
5791f63
Add pkgdown reference to CsvWriteOptions
thisisnic Apr 22, 2021
deff496
Add WriteOptions to arrow::csv namespace
thisisnic Apr 23, 2021
e21aaa5
Update types
thisisnic Apr 23, 2021
04d8106
Update arrowExports
thisisnic Apr 23, 2021
59112f6
Re-order params and add assertion
thisisnic Apr 23, 2021
c172db8
try changing
thisisnic Apr 23, 2021
2b962ad
Remove const keyword
thisisnic Apr 23, 2021
1d70f0b
Remove unnecessary comma, include relevant header files, refactor cpp
thisisnic Apr 23, 2021
713075f
Remove exposed memory pool argument, update NAMESPACE and docs
thisisnic Apr 26, 2021
86d83d8
Use gc_memory_pool() instead of arrow::default_memory_pool() to preve…
thisisnic Apr 26, 2021
8de7c3b
Typo fix
thisisnic Apr 26, 2021
3239575
Skip tests that include writing date columns
thisisnic Apr 27, 2021
63a2f98
Change whitespace to force CI
thisisnic Apr 27, 2021
96e9e2e
Change R C++ function format
thisisnic Apr 27, 2021
fcc6f5f
Run linter on R C++
thisisnic Apr 27, 2021
ee38464
Update docs
thisisnic Apr 27, 2021
6b6914c
Add write_csv_arrrow to _pkgdown.yml
thisisnic Apr 28, 2021
93338cc
Inconsequential grammar change to trigger CI
thisisnic Apr 28, 2021
7722206
Remove extra no-dates tests
thisisnic Apr 30, 2021
0d62fa1
Move docs for CsvWriteOptions
thisisnic Apr 30, 2021
faab7ff
Move tests to bottom of file
thisisnic May 3, 2021
8ff781f
Add tests for invalid inputs
thisisnic May 3, 2021
a421243
Remove extra whitespace
thisisnic May 3, 2021
6252dcd
Move batch_size validation
nealrichardson May 4, 2021
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions cpp/src/arrow/csv/type_fwd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -22,6 +22,7 @@ class TableReader;
struct ConvertOptions;
struct ReadOptions;
struct ParseOptions;
struct WriteOptions;

} // namespace csv
} // namespace arrow
2 changes: 2 additions & 0 deletions r/NAMESPACE
Original file line numberDiff line numberDiff line change
Expand Up@@ -122,6 +122,7 @@ export(CsvFragmentScanOptions)
export(CsvParseOptions)
export(CsvReadOptions)
export(CsvTableReader)
export(CsvWriteOptions)
export(Dataset)
export(DatasetFactory)
export(DateUnit)
Expand DownExpand Up@@ -277,6 +278,7 @@ export(unify_schemas)
export(utf8)
export(value_counts)
export(write_arrow)
export(write_csv_arrow)
export(write_dataset)
export(write_feather)
export(write_ipc_stream)
Expand Down
12 changes: 12 additions & 0 deletions r/R/arrowExports.R

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

64 changes: 64 additions & 0 deletions r/R/csv.R
Original file line numberDiff line numberDiff line change
Expand Up@@ -381,6 +381,11 @@ CsvTableReader$create <- function(file,
#' `TimestampParser$create()` takes an optional `format` string argument.
#' See [`strptime()`][base::strptime()] for example syntax.
#' The default is to use an ISO-8601 format parser.
#'
#' The `CsvWriteOptions$create()` factory method takes the following arguments:
#' - `include_header` Whether to write an initial header line with column names
#' - `batch_size` Maximum number of rows processed at a time. Default is 1024.
#'
#' @section Active bindings:
#'
#' - `column_names`: from `CsvReadOptions`
Expand DownExpand Up@@ -408,6 +413,19 @@ CsvReadOptions$create <- function(use_threads = option_use_threads(),
)
}

#' @rdname CsvReadOptions
#' @export
CsvWriteOptions <- R6Class("CsvWriteOptions", inherit = ArrowObject)
CsvWriteOptions$create <- function(include_header = TRUE, batch_size = 1024L){
assert_that(is_integerish(batch_size, n = 1, finite = TRUE), batch_size > 0)
csv___WriteOptions__initialize(
list(
include_header = include_header,
batch_size = as.integer(batch_size)
)
)
}

readr_to_csv_read_options <- function(skip, col_names, col_types) {
if (isTRUE(col_names)) {
# C++ default to parse is 0-length string array
Expand DownExpand Up@@ -585,3 +603,49 @@ readr_to_csv_convert_options <- function(na,
include_columns = include_columns
)
}

#' Write CSV file to disk
#'
#' @param x `data.frame`, [RecordBatch], or [Table]
#' @param sink A string file path, URI, or [OutputStream], or path in a file
#' system (`SubTreeFileSystem`)
#' @param include_header Whether to write an initial header line with column names
#' @param batch_size Maximum number of rows processed at a time. Default is 1024.
#'
#' @return The input `x`, invisibly. Note that if `sink` is an [OutputStream],
#' the stream will be left open.
#' @export
#' @examples
#' \donttest{
#' tf <- tempfile()
#' on.exit(unlink(tf))
#' write_csv_arrow(mtcars, tf)
#' }
#' @include arrow-package.R
write_csv_arrow <- function(x,
sink,
include_header = TRUE,
batch_size = 1024L) {

write_options <- CsvWriteOptions$create(include_header, batch_size)

x_out <- x
if (is.data.frame(x)) {
x <- Table$create(x)
}

assert_is(x, "ArrowTabular")

if (!inherits(sink, "OutputStream")) {
sink <- make_output_stream(sink)
on.exit(sink$close())
}

if(inherits(x, "RecordBatch")){
csv___WriteCSV__RecordBatch(x, write_options, sink)
} else if(inherits(x, "Table")){
csv___WriteCSV__Table(x, write_options, sink)
}

invisible(x_out)
}
2 changes: 2 additions & 0 deletions r/_pkgdown.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -98,6 +98,7 @@ reference:
- write_ipc_stream
- write_to_raw
- write_parquet
- write_csv_arrow
- title: C++ reader/writer interface
contents:
- ParquetFileReader
Expand All@@ -109,6 +110,7 @@ reference:
- RecordBatchReader
- RecordBatchWriter
- CsvReadOptions
- CsvWriteOptions
- title: Arrow data containers
contents:
- array
Expand Down
22 changes: 22 additions & 0 deletions r/man/CsvWriteOptions.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

32 changes: 32 additions & 0 deletions r/man/write_csv_arrow.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

54 changes: 54 additions & 0 deletions r/src/arrowExports.cpp

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions r/src/arrow_types.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -178,6 +178,7 @@ R6_CLASS_NAME(arrow::csv::ReadOptions, "CsvReadOptions");
R6_CLASS_NAME(arrow::csv::ParseOptions, "CsvParseOptions");
R6_CLASS_NAME(arrow::csv::ConvertOptions, "CsvConvertOptions");
R6_CLASS_NAME(arrow::csv::TableReader, "CsvTableReader");
R6_CLASS_NAME(arrow::csv::WriteOptions, "CsvWriteOptions");

#if defined(ARROW_R_WITH_PARQUET)
R6_CLASS_NAME(parquet::ArrowReaderProperties, "ParquetArrowReaderProperties");
Expand Down
30 changes: 30 additions & 0 deletions r/src/csv.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,8 +20,21 @@
#if defined(ARROW_R_WITH_ARROW)

#include <arrow/csv/reader.h>
#include <arrow/csv/writer.h>
#include <arrow/memory_pool.h>

#include <arrow/util/value_parsing.h>

// [[arrow::export]]
std::shared_ptr<arrow::csv::WriteOptions> csv___WriteOptions__initialize(
cpp11::list options) {
auto res =
std::make_shared<arrow::csv::WriteOptions>(arrow::csv::WriteOptions::Defaults());
res->include_header = cpp11::as_cpp<bool>(options["include_header"]);
res->batch_size = cpp11::as_cpp<int>(options["batch_size"]);
return res;
}

// [[arrow::export]]
std::shared_ptr<arrow::csv::ReadOptions> csv___ReadOptions__initialize(
cpp11::list options) {
Expand DownExpand Up@@ -174,4 +187,21 @@ std::shared_ptr<arrow::TimestampParser> TimestampParser__MakeISO8601() {
return arrow::TimestampParser::MakeISO8601();
}

// [[arrow::export]]
void csv___WriteCSV__Table(const std::shared_ptr<arrow::Table>& table,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(
arrow::csv::WriteCSV(*table, *write_options, gc_memory_pool(), stream.get()));
}

// [[arrow::export]]
void csv___WriteCSV__RecordBatch(
const std::shared_ptr<arrow::RecordBatch>& record_batch,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(arrow::csv::WriteCSV(*record_batch, *write_options, gc_memory_pool(),
stream.get()));
}

#endif
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
b4df3e7
Add unit tests for write_csv_arrow
thisisnic Apr 22, 2021
1077e8d
Add CsvWriteOptions and write_csv_arrow functions
thisisnic Apr 22, 2021
9d12fe6
Add C++ functions for intialising csv writeoptions and bindings to ca…
thisisnic Apr 22, 2021
5791f63
Add pkgdown reference to CsvWriteOptions
thisisnic Apr 22, 2021
deff496
Add WriteOptions to arrow::csv namespace
thisisnic Apr 23, 2021
e21aaa5
Update types
thisisnic Apr 23, 2021
04d8106
Update arrowExports
thisisnic Apr 23, 2021
59112f6
Re-order params and add assertion
thisisnic Apr 23, 2021
c172db8
try changing
thisisnic Apr 23, 2021
2b962ad
Remove const keyword
thisisnic Apr 23, 2021
1d70f0b
Remove unnecessary comma, include relevant header files, refactor cpp
thisisnic Apr 23, 2021
713075f
Remove exposed memory pool argument, update NAMESPACE and docs
thisisnic Apr 26, 2021
86d83d8
Use gc_memory_pool() instead of arrow::default_memory_pool() to preve…
thisisnic Apr 26, 2021
8de7c3b
Typo fix
thisisnic Apr 26, 2021
3239575
Skip tests that include writing date columns
thisisnic Apr 27, 2021
63a2f98
Change whitespace to force CI
thisisnic Apr 27, 2021
96e9e2e
Change R C++ function format
thisisnic Apr 27, 2021
fcc6f5f
Run linter on R C++
thisisnic Apr 27, 2021
ee38464
Update docs
thisisnic Apr 27, 2021
6b6914c
Add write_csv_arrrow to _pkgdown.yml
thisisnic Apr 28, 2021
93338cc
Inconsequential grammar change to trigger CI
thisisnic Apr 28, 2021
7722206
Remove extra no-dates tests
thisisnic Apr 30, 2021
0d62fa1
Move docs for CsvWriteOptions
thisisnic Apr 30, 2021
faab7ff
Move tests to bottom of file
thisisnic May 3, 2021
8ff781f
Add tests for invalid inputs
thisisnic May 3, 2021
a421243
Remove extra whitespace
thisisnic May 3, 2021
6252dcd
Move batch_size validation
nealrichardson May 4, 2021
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions cpp/src/arrow/csv/type_fwd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -22,6 +22,7 @@ class TableReader;
struct ConvertOptions;
struct ReadOptions;
struct ParseOptions;
struct WriteOptions;

} // namespace csv
} // namespace arrow
2 changes: 2 additions & 0 deletions r/NAMESPACE
Original file line numberDiff line numberDiff line change
Expand Up@@ -122,6 +122,7 @@ export(CsvFragmentScanOptions)
export(CsvParseOptions)
export(CsvReadOptions)
export(CsvTableReader)
export(CsvWriteOptions)
export(Dataset)
export(DatasetFactory)
export(DateUnit)
Expand DownExpand Up@@ -277,6 +278,7 @@ export(unify_schemas)
export(utf8)
export(value_counts)
export(write_arrow)
export(write_csv_arrow)
export(write_dataset)
export(write_feather)
export(write_ipc_stream)
Expand Down
12 changes: 12 additions & 0 deletions r/R/arrowExports.R

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

64 changes: 64 additions & 0 deletions r/R/csv.R
Original file line numberDiff line numberDiff line change
Expand Up@@ -381,6 +381,11 @@ CsvTableReader$create <- function(file,
#' `TimestampParser$create()` takes an optional `format` string argument.
#' See [`strptime()`][base::strptime()] for example syntax.
#' The default is to use an ISO-8601 format parser.
#'
#' The `CsvWriteOptions$create()` factory method takes the following arguments:
#' - `include_header` Whether to write an initial header line with column names
#' - `batch_size` Maximum number of rows processed at a time. Default is 1024.
#'
#' @section Active bindings:
#'
#' - `column_names`: from `CsvReadOptions`
Expand DownExpand Up@@ -408,6 +413,19 @@ CsvReadOptions$create <- function(use_threads = option_use_threads(),
)
}

#' @rdname CsvReadOptions
#' @export
CsvWriteOptions <- R6Class("CsvWriteOptions", inherit = ArrowObject)
CsvWriteOptions$create <- function(include_header = TRUE, batch_size = 1024L){
assert_that(is_integerish(batch_size, n = 1, finite = TRUE), batch_size > 0)
csv___WriteOptions__initialize(
list(
include_header = include_header,
batch_size = as.integer(batch_size)
)
)
}

readr_to_csv_read_options <- function(skip, col_names, col_types) {
if (isTRUE(col_names)) {
# C++ default to parse is 0-length string array
Expand DownExpand Up@@ -585,3 +603,49 @@ readr_to_csv_convert_options <- function(na,
include_columns = include_columns
)
}

#' Write CSV file to disk
#'
#' @param x `data.frame`, [RecordBatch], or [Table]
#' @param sink A string file path, URI, or [OutputStream], or path in a file
#' system (`SubTreeFileSystem`)
#' @param include_header Whether to write an initial header line with column names
#' @param batch_size Maximum number of rows processed at a time. Default is 1024.
#'
#' @return The input `x`, invisibly. Note that if `sink` is an [OutputStream],
#' the stream will be left open.
#' @export
#' @examples
#' \donttest{
#' tf <- tempfile()
#' on.exit(unlink(tf))
#' write_csv_arrow(mtcars, tf)
#' }
#' @include arrow-package.R
write_csv_arrow <- function(x,
sink,
include_header = TRUE,
batch_size = 1024L) {

write_options <- CsvWriteOptions$create(include_header, batch_size)

x_out <- x
if (is.data.frame(x)) {
x <- Table$create(x)
}

assert_is(x, "ArrowTabular")

if (!inherits(sink, "OutputStream")) {
sink <- make_output_stream(sink)
on.exit(sink$close())
}

if(inherits(x, "RecordBatch")){
csv___WriteCSV__RecordBatch(x, write_options, sink)
} else if(inherits(x, "Table")){
csv___WriteCSV__Table(x, write_options, sink)
}

invisible(x_out)
}
2 changes: 2 additions & 0 deletions r/_pkgdown.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -98,6 +98,7 @@ reference:
- write_ipc_stream
- write_to_raw
- write_parquet
- write_csv_arrow
- title: C++ reader/writer interface
contents:
- ParquetFileReader
Expand All@@ -109,6 +110,7 @@ reference:
- RecordBatchReader
- RecordBatchWriter
- CsvReadOptions
- CsvWriteOptions
- title: Arrow data containers
contents:
- array
Expand Down
22 changes: 22 additions & 0 deletions r/man/CsvWriteOptions.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

32 changes: 32 additions & 0 deletions r/man/write_csv_arrow.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

54 changes: 54 additions & 0 deletions r/src/arrowExports.cpp

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions r/src/arrow_types.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -178,6 +178,7 @@ R6_CLASS_NAME(arrow::csv::ReadOptions, "CsvReadOptions");
R6_CLASS_NAME(arrow::csv::ParseOptions, "CsvParseOptions");
R6_CLASS_NAME(arrow::csv::ConvertOptions, "CsvConvertOptions");
R6_CLASS_NAME(arrow::csv::TableReader, "CsvTableReader");
R6_CLASS_NAME(arrow::csv::WriteOptions, "CsvWriteOptions");

#if defined(ARROW_R_WITH_PARQUET)
R6_CLASS_NAME(parquet::ArrowReaderProperties, "ParquetArrowReaderProperties");
Expand Down
30 changes: 30 additions & 0 deletions r/src/csv.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,8 +20,21 @@
#if defined(ARROW_R_WITH_ARROW)

#include <arrow/csv/reader.h>
#include <arrow/csv/writer.h>
#include <arrow/memory_pool.h>

#include <arrow/util/value_parsing.h>

// [[arrow::export]]
std::shared_ptr<arrow::csv::WriteOptions> csv___WriteOptions__initialize(
cpp11::list options) {
auto res =
std::make_shared<arrow::csv::WriteOptions>(arrow::csv::WriteOptions::Defaults());
res->include_header = cpp11::as_cpp<bool>(options["include_header"]);
res->batch_size = cpp11::as_cpp<int>(options["batch_size"]);
return res;
}

// [[arrow::export]]
std::shared_ptr<arrow::csv::ReadOptions> csv___ReadOptions__initialize(
cpp11::list options) {
Expand DownExpand Up@@ -174,4 +187,21 @@ std::shared_ptr<arrow::TimestampParser> TimestampParser__MakeISO8601() {
return arrow::TimestampParser::MakeISO8601();
}

// [[arrow::export]]
void csv___WriteCSV__Table(const std::shared_ptr<arrow::Table>& table,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(
arrow::csv::WriteCSV(*table, *write_options, gc_memory_pool(), stream.get()));
}

// [[arrow::export]]
void csv___WriteCSV__RecordBatch(
const std::shared_ptr<arrow::RecordBatch>& record_batch,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(arrow::csv::WriteCSV(*record_batch, *write_options, gc_memory_pool(),
stream.get()));
}

#endif
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
b4df3e7
Add unit tests for write_csv_arrow
thisisnic Apr 22, 2021
1077e8d
Add CsvWriteOptions and write_csv_arrow functions
thisisnic Apr 22, 2021
9d12fe6
Add C++ functions for intialising csv writeoptions and bindings to ca…
thisisnic Apr 22, 2021
5791f63
Add pkgdown reference to CsvWriteOptions
thisisnic Apr 22, 2021
deff496
Add WriteOptions to arrow::csv namespace
thisisnic Apr 23, 2021
e21aaa5
Update types
thisisnic Apr 23, 2021
04d8106
Update arrowExports
thisisnic Apr 23, 2021
59112f6
Re-order params and add assertion
thisisnic Apr 23, 2021
c172db8
try changing
thisisnic Apr 23, 2021
2b962ad
Remove const keyword
thisisnic Apr 23, 2021
1d70f0b
Remove unnecessary comma, include relevant header files, refactor cpp
thisisnic Apr 23, 2021
713075f
Remove exposed memory pool argument, update NAMESPACE and docs
thisisnic Apr 26, 2021
86d83d8
Use gc_memory_pool() instead of arrow::default_memory_pool() to preve…
thisisnic Apr 26, 2021
8de7c3b
Typo fix
thisisnic Apr 26, 2021
3239575
Skip tests that include writing date columns
thisisnic Apr 27, 2021
63a2f98
Change whitespace to force CI
thisisnic Apr 27, 2021
96e9e2e
Change R C++ function format
thisisnic Apr 27, 2021
fcc6f5f
Run linter on R C++
thisisnic Apr 27, 2021
ee38464
Update docs
thisisnic Apr 27, 2021
6b6914c
Add write_csv_arrrow to _pkgdown.yml
thisisnic Apr 28, 2021
93338cc
Inconsequential grammar change to trigger CI
thisisnic Apr 28, 2021
7722206
Remove extra no-dates tests
thisisnic Apr 30, 2021
0d62fa1
Move docs for CsvWriteOptions
thisisnic Apr 30, 2021
faab7ff
Move tests to bottom of file
thisisnic May 3, 2021
8ff781f
Add tests for invalid inputs
thisisnic May 3, 2021
a421243
Remove extra whitespace
thisisnic May 3, 2021
6252dcd
Move batch_size validation
nealrichardson May 4, 2021
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions cpp/src/arrow/csv/type_fwd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -22,6 +22,7 @@ class TableReader;
struct ConvertOptions;
struct ReadOptions;
struct ParseOptions;
struct WriteOptions;

} // namespace csv
} // namespace arrow
2 changes: 2 additions & 0 deletions r/NAMESPACE
Original file line numberDiff line numberDiff line change
Expand Up@@ -122,6 +122,7 @@ export(CsvFragmentScanOptions)
export(CsvParseOptions)
export(CsvReadOptions)
export(CsvTableReader)
export(CsvWriteOptions)
export(Dataset)
export(DatasetFactory)
export(DateUnit)
Expand DownExpand Up@@ -277,6 +278,7 @@ export(unify_schemas)
export(utf8)
export(value_counts)
export(write_arrow)
export(write_csv_arrow)
export(write_dataset)
export(write_feather)
export(write_ipc_stream)
Expand Down
12 changes: 12 additions & 0 deletions r/R/arrowExports.R

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

64 changes: 64 additions & 0 deletions r/R/csv.R
Original file line numberDiff line numberDiff line change
Expand Up@@ -381,6 +381,11 @@ CsvTableReader$create <- function(file,
#' `TimestampParser$create()` takes an optional `format` string argument.
#' See [`strptime()`][base::strptime()] for example syntax.
#' The default is to use an ISO-8601 format parser.
#'
#' The `CsvWriteOptions$create()` factory method takes the following arguments:
#' - `include_header` Whether to write an initial header line with column names
#' - `batch_size` Maximum number of rows processed at a time. Default is 1024.
#'
#' @section Active bindings:
#'
#' - `column_names`: from `CsvReadOptions`
Expand DownExpand Up@@ -408,6 +413,19 @@ CsvReadOptions$create <- function(use_threads = option_use_threads(),
)
}

#' @rdname CsvReadOptions
#' @export
CsvWriteOptions <- R6Class("CsvWriteOptions", inherit = ArrowObject)
CsvWriteOptions$create <- function(include_header = TRUE, batch_size = 1024L){
assert_that(is_integerish(batch_size, n = 1, finite = TRUE), batch_size > 0)
csv___WriteOptions__initialize(
list(
include_header = include_header,
batch_size = as.integer(batch_size)
)
)
}

readr_to_csv_read_options <- function(skip, col_names, col_types) {
if (isTRUE(col_names)) {
# C++ default to parse is 0-length string array
Expand DownExpand Up@@ -585,3 +603,49 @@ readr_to_csv_convert_options <- function(na,
include_columns = include_columns
)
}

#' Write CSV file to disk
#'
#' @param x `data.frame`, [RecordBatch], or [Table]
#' @param sink A string file path, URI, or [OutputStream], or path in a file
#' system (`SubTreeFileSystem`)
#' @param include_header Whether to write an initial header line with column names
#' @param batch_size Maximum number of rows processed at a time. Default is 1024.
#'
#' @return The input `x`, invisibly. Note that if `sink` is an [OutputStream],
#' the stream will be left open.
#' @export
#' @examples
#' \donttest{
#' tf <- tempfile()
#' on.exit(unlink(tf))
#' write_csv_arrow(mtcars, tf)
#' }
#' @include arrow-package.R
write_csv_arrow <- function(x,
sink,
include_header = TRUE,
batch_size = 1024L) {

write_options <- CsvWriteOptions$create(include_header, batch_size)

x_out <- x
if (is.data.frame(x)) {
x <- Table$create(x)
}

assert_is(x, "ArrowTabular")

if (!inherits(sink, "OutputStream")) {
sink <- make_output_stream(sink)
on.exit(sink$close())
}

if(inherits(x, "RecordBatch")){
csv___WriteCSV__RecordBatch(x, write_options, sink)
} else if(inherits(x, "Table")){
csv___WriteCSV__Table(x, write_options, sink)
}

invisible(x_out)
}
2 changes: 2 additions & 0 deletions r/_pkgdown.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -98,6 +98,7 @@ reference:
- write_ipc_stream
- write_to_raw
- write_parquet
- write_csv_arrow
- title: C++ reader/writer interface
contents:
- ParquetFileReader
Expand All@@ -109,6 +110,7 @@ reference:
- RecordBatchReader
- RecordBatchWriter
- CsvReadOptions
- CsvWriteOptions
- title: Arrow data containers
contents:
- array
Expand Down
22 changes: 22 additions & 0 deletions r/man/CsvWriteOptions.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

32 changes: 32 additions & 0 deletions r/man/write_csv_arrow.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

54 changes: 54 additions & 0 deletions r/src/arrowExports.cpp

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions r/src/arrow_types.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -178,6 +178,7 @@ R6_CLASS_NAME(arrow::csv::ReadOptions, "CsvReadOptions");
R6_CLASS_NAME(arrow::csv::ParseOptions, "CsvParseOptions");
R6_CLASS_NAME(arrow::csv::ConvertOptions, "CsvConvertOptions");
R6_CLASS_NAME(arrow::csv::TableReader, "CsvTableReader");
R6_CLASS_NAME(arrow::csv::WriteOptions, "CsvWriteOptions");

#if defined(ARROW_R_WITH_PARQUET)
R6_CLASS_NAME(parquet::ArrowReaderProperties, "ParquetArrowReaderProperties");
Expand Down
30 changes: 30 additions & 0 deletions r/src/csv.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,8 +20,21 @@
#if defined(ARROW_R_WITH_ARROW)

#include <arrow/csv/reader.h>
#include <arrow/csv/writer.h>
#include <arrow/memory_pool.h>

#include <arrow/util/value_parsing.h>

// [[arrow::export]]
std::shared_ptr<arrow::csv::WriteOptions> csv___WriteOptions__initialize(
cpp11::list options) {
auto res =
std::make_shared<arrow::csv::WriteOptions>(arrow::csv::WriteOptions::Defaults());
res->include_header = cpp11::as_cpp<bool>(options["include_header"]);
res->batch_size = cpp11::as_cpp<int>(options["batch_size"]);
return res;
}

// [[arrow::export]]
std::shared_ptr<arrow::csv::ReadOptions> csv___ReadOptions__initialize(
cpp11::list options) {
Expand DownExpand Up@@ -174,4 +187,21 @@ std::shared_ptr<arrow::TimestampParser> TimestampParser__MakeISO8601() {
return arrow::TimestampParser::MakeISO8601();
}

// [[arrow::export]]
void csv___WriteCSV__Table(const std::shared_ptr<arrow::Table>& table,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(
arrow::csv::WriteCSV(*table, *write_options, gc_memory_pool(), stream.get()));
}

// [[arrow::export]]
void csv___WriteCSV__RecordBatch(
const std::shared_ptr<arrow::RecordBatch>& record_batch,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(arrow::csv::WriteCSV(*record_batch, *write_options, gc_memory_pool(),
stream.get()));
}

#endif
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
b4df3e7
Add unit tests for write_csv_arrow
thisisnic Apr 22, 2021
1077e8d
Add CsvWriteOptions and write_csv_arrow functions
thisisnic Apr 22, 2021
9d12fe6
Add C++ functions for intialising csv writeoptions and bindings to ca…
thisisnic Apr 22, 2021
5791f63
Add pkgdown reference to CsvWriteOptions
thisisnic Apr 22, 2021
deff496
Add WriteOptions to arrow::csv namespace
thisisnic Apr 23, 2021
e21aaa5
Update types
thisisnic Apr 23, 2021
04d8106
Update arrowExports
thisisnic Apr 23, 2021
59112f6
Re-order params and add assertion
thisisnic Apr 23, 2021
c172db8
try changing
thisisnic Apr 23, 2021
2b962ad
Remove const keyword
thisisnic Apr 23, 2021
1d70f0b
Remove unnecessary comma, include relevant header files, refactor cpp
thisisnic Apr 23, 2021
713075f
Remove exposed memory pool argument, update NAMESPACE and docs
thisisnic Apr 26, 2021
86d83d8
Use gc_memory_pool() instead of arrow::default_memory_pool() to preve…
thisisnic Apr 26, 2021
8de7c3b
Typo fix
thisisnic Apr 26, 2021
3239575
Skip tests that include writing date columns
thisisnic Apr 27, 2021
63a2f98
Change whitespace to force CI
thisisnic Apr 27, 2021
96e9e2e
Change R C++ function format
thisisnic Apr 27, 2021
fcc6f5f
Run linter on R C++
thisisnic Apr 27, 2021
ee38464
Update docs
thisisnic Apr 27, 2021
6b6914c
Add write_csv_arrrow to _pkgdown.yml
thisisnic Apr 28, 2021
93338cc
Inconsequential grammar change to trigger CI
thisisnic Apr 28, 2021
7722206
Remove extra no-dates tests
thisisnic Apr 30, 2021
0d62fa1
Move docs for CsvWriteOptions
thisisnic Apr 30, 2021
faab7ff
Move tests to bottom of file
thisisnic May 3, 2021
8ff781f
Add tests for invalid inputs
thisisnic May 3, 2021
a421243
Remove extra whitespace
thisisnic May 3, 2021
6252dcd
Move batch_size validation
nealrichardson May 4, 2021
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions cpp/src/arrow/csv/type_fwd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -22,6 +22,7 @@ class TableReader;
struct ConvertOptions;
struct ReadOptions;
struct ParseOptions;
struct WriteOptions;

} // namespace csv
} // namespace arrow
2 changes: 2 additions & 0 deletions r/NAMESPACE
Original file line numberDiff line numberDiff line change
Expand Up@@ -122,6 +122,7 @@ export(CsvFragmentScanOptions)
export(CsvParseOptions)
export(CsvReadOptions)
export(CsvTableReader)
export(CsvWriteOptions)
export(Dataset)
export(DatasetFactory)
export(DateUnit)
Expand DownExpand Up@@ -277,6 +278,7 @@ export(unify_schemas)
export(utf8)
export(value_counts)
export(write_arrow)
export(write_csv_arrow)
export(write_dataset)
export(write_feather)
export(write_ipc_stream)
Expand Down
12 changes: 12 additions & 0 deletions r/R/arrowExports.R

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

64 changes: 64 additions & 0 deletions r/R/csv.R
Original file line numberDiff line numberDiff line change
Expand Up@@ -381,6 +381,11 @@ CsvTableReader$create <- function(file,
#' `TimestampParser$create()` takes an optional `format` string argument.
#' See [`strptime()`][base::strptime()] for example syntax.
#' The default is to use an ISO-8601 format parser.
#'
#' The `CsvWriteOptions$create()` factory method takes the following arguments:
#' - `include_header` Whether to write an initial header line with column names
#' - `batch_size` Maximum number of rows processed at a time. Default is 1024.
#'
#' @section Active bindings:
#'
#' - `column_names`: from `CsvReadOptions`
Expand DownExpand Up@@ -408,6 +413,19 @@ CsvReadOptions$create <- function(use_threads = option_use_threads(),
)
}

#' @rdname CsvReadOptions
#' @export
CsvWriteOptions <- R6Class("CsvWriteOptions", inherit = ArrowObject)
CsvWriteOptions$create <- function(include_header = TRUE, batch_size = 1024L){
assert_that(is_integerish(batch_size, n = 1, finite = TRUE), batch_size > 0)
csv___WriteOptions__initialize(
list(
include_header = include_header,
batch_size = as.integer(batch_size)
)
)
}

readr_to_csv_read_options <- function(skip, col_names, col_types) {
if (isTRUE(col_names)) {
# C++ default to parse is 0-length string array
Expand DownExpand Up@@ -585,3 +603,49 @@ readr_to_csv_convert_options <- function(na,
include_columns = include_columns
)
}

#' Write CSV file to disk
#'
#' @param x `data.frame`, [RecordBatch], or [Table]
#' @param sink A string file path, URI, or [OutputStream], or path in a file
#' system (`SubTreeFileSystem`)
#' @param include_header Whether to write an initial header line with column names
#' @param batch_size Maximum number of rows processed at a time. Default is 1024.
#'
#' @return The input `x`, invisibly. Note that if `sink` is an [OutputStream],
#' the stream will be left open.
#' @export
#' @examples
#' \donttest{
#' tf <- tempfile()
#' on.exit(unlink(tf))
#' write_csv_arrow(mtcars, tf)
#' }
#' @include arrow-package.R
write_csv_arrow <- function(x,
sink,
include_header = TRUE,
batch_size = 1024L) {

write_options <- CsvWriteOptions$create(include_header, batch_size)

x_out <- x
if (is.data.frame(x)) {
x <- Table$create(x)
}

assert_is(x, "ArrowTabular")

if (!inherits(sink, "OutputStream")) {
sink <- make_output_stream(sink)
on.exit(sink$close())
}

if(inherits(x, "RecordBatch")){
csv___WriteCSV__RecordBatch(x, write_options, sink)
} else if(inherits(x, "Table")){
csv___WriteCSV__Table(x, write_options, sink)
}

invisible(x_out)
}
2 changes: 2 additions & 0 deletions r/_pkgdown.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -98,6 +98,7 @@ reference:
- write_ipc_stream
- write_to_raw
- write_parquet
- write_csv_arrow
- title: C++ reader/writer interface
contents:
- ParquetFileReader
Expand All@@ -109,6 +110,7 @@ reference:
- RecordBatchReader
- RecordBatchWriter
- CsvReadOptions
- CsvWriteOptions
- title: Arrow data containers
contents:
- array
Expand Down
22 changes: 22 additions & 0 deletions r/man/CsvWriteOptions.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

32 changes: 32 additions & 0 deletions r/man/write_csv_arrow.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

54 changes: 54 additions & 0 deletions r/src/arrowExports.cpp

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions r/src/arrow_types.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -178,6 +178,7 @@ R6_CLASS_NAME(arrow::csv::ReadOptions, "CsvReadOptions");
R6_CLASS_NAME(arrow::csv::ParseOptions, "CsvParseOptions");
R6_CLASS_NAME(arrow::csv::ConvertOptions, "CsvConvertOptions");
R6_CLASS_NAME(arrow::csv::TableReader, "CsvTableReader");
R6_CLASS_NAME(arrow::csv::WriteOptions, "CsvWriteOptions");

#if defined(ARROW_R_WITH_PARQUET)
R6_CLASS_NAME(parquet::ArrowReaderProperties, "ParquetArrowReaderProperties");
Expand Down
30 changes: 30 additions & 0 deletions r/src/csv.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,8 +20,21 @@
#if defined(ARROW_R_WITH_ARROW)

#include <arrow/csv/reader.h>
#include <arrow/csv/writer.h>
#include <arrow/memory_pool.h>

#include <arrow/util/value_parsing.h>

// [[arrow::export]]
std::shared_ptr<arrow::csv::WriteOptions> csv___WriteOptions__initialize(
cpp11::list options) {
auto res =
std::make_shared<arrow::csv::WriteOptions>(arrow::csv::WriteOptions::Defaults());
res->include_header = cpp11::as_cpp<bool>(options["include_header"]);
res->batch_size = cpp11::as_cpp<int>(options["batch_size"]);
return res;
}

// [[arrow::export]]
std::shared_ptr<arrow::csv::ReadOptions> csv___ReadOptions__initialize(
cpp11::list options) {
Expand DownExpand Up@@ -174,4 +187,21 @@ std::shared_ptr<arrow::TimestampParser> TimestampParser__MakeISO8601() {
return arrow::TimestampParser::MakeISO8601();
}

// [[arrow::export]]
void csv___WriteCSV__Table(const std::shared_ptr<arrow::Table>& table,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(
arrow::csv::WriteCSV(*table, *write_options, gc_memory_pool(), stream.get()));
}

// [[arrow::export]]
void csv___WriteCSV__RecordBatch(
const std::shared_ptr<arrow::RecordBatch>& record_batch,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(arrow::csv::WriteCSV(*record_batch, *write_options, gc_memory_pool(),
stream.get()));
}

#endif
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
b4df3e7
Add unit tests for write_csv_arrow
thisisnic Apr 22, 2021
1077e8d
Add CsvWriteOptions and write_csv_arrow functions
thisisnic Apr 22, 2021
9d12fe6
Add C++ functions for intialising csv writeoptions and bindings to ca…
thisisnic Apr 22, 2021
5791f63
Add pkgdown reference to CsvWriteOptions
thisisnic Apr 22, 2021
deff496
Add WriteOptions to arrow::csv namespace
thisisnic Apr 23, 2021
e21aaa5
Update types
thisisnic Apr 23, 2021
04d8106
Update arrowExports
thisisnic Apr 23, 2021
59112f6
Re-order params and add assertion
thisisnic Apr 23, 2021
c172db8
try changing
thisisnic Apr 23, 2021
2b962ad
Remove const keyword
thisisnic Apr 23, 2021
1d70f0b
Remove unnecessary comma, include relevant header files, refactor cpp
thisisnic Apr 23, 2021
713075f
Remove exposed memory pool argument, update NAMESPACE and docs
thisisnic Apr 26, 2021
86d83d8
Use gc_memory_pool() instead of arrow::default_memory_pool() to preve…
thisisnic Apr 26, 2021
8de7c3b
Typo fix
thisisnic Apr 26, 2021
3239575
Skip tests that include writing date columns
thisisnic Apr 27, 2021
63a2f98
Change whitespace to force CI
thisisnic Apr 27, 2021
96e9e2e
Change R C++ function format
thisisnic Apr 27, 2021
fcc6f5f
Run linter on R C++
thisisnic Apr 27, 2021
ee38464
Update docs
thisisnic Apr 27, 2021
6b6914c
Add write_csv_arrrow to _pkgdown.yml
thisisnic Apr 28, 2021
93338cc
Inconsequential grammar change to trigger CI
thisisnic Apr 28, 2021
7722206
Remove extra no-dates tests
thisisnic Apr 30, 2021
0d62fa1
Move docs for CsvWriteOptions
thisisnic Apr 30, 2021
faab7ff
Move tests to bottom of file
thisisnic May 3, 2021
8ff781f
Add tests for invalid inputs
thisisnic May 3, 2021
a421243
Remove extra whitespace
thisisnic May 3, 2021
6252dcd
Move batch_size validation
nealrichardson May 4, 2021
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions cpp/src/arrow/csv/type_fwd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -22,6 +22,7 @@ class TableReader;
struct ConvertOptions;
struct ReadOptions;
struct ParseOptions;
struct WriteOptions;

} // namespace csv
} // namespace arrow
2 changes: 2 additions & 0 deletions r/NAMESPACE
Original file line numberDiff line numberDiff line change
Expand Up@@ -122,6 +122,7 @@ export(CsvFragmentScanOptions)
export(CsvParseOptions)
export(CsvReadOptions)
export(CsvTableReader)
export(CsvWriteOptions)
export(Dataset)
export(DatasetFactory)
export(DateUnit)
Expand DownExpand Up@@ -277,6 +278,7 @@ export(unify_schemas)
export(utf8)
export(value_counts)
export(write_arrow)
export(write_csv_arrow)
export(write_dataset)
export(write_feather)
export(write_ipc_stream)
Expand Down
12 changes: 12 additions & 0 deletions r/R/arrowExports.R

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

64 changes: 64 additions & 0 deletions r/R/csv.R
Original file line numberDiff line numberDiff line change
Expand Up@@ -381,6 +381,11 @@ CsvTableReader$create <- function(file,
#' `TimestampParser$create()` takes an optional `format` string argument.
#' See [`strptime()`][base::strptime()] for example syntax.
#' The default is to use an ISO-8601 format parser.
#'
#' The `CsvWriteOptions$create()` factory method takes the following arguments:
#' - `include_header` Whether to write an initial header line with column names
#' - `batch_size` Maximum number of rows processed at a time. Default is 1024.
#'
#' @section Active bindings:
#'
#' - `column_names`: from `CsvReadOptions`
Expand DownExpand Up@@ -408,6 +413,19 @@ CsvReadOptions$create <- function(use_threads = option_use_threads(),
)
}

#' @rdname CsvReadOptions
#' @export
CsvWriteOptions <- R6Class("CsvWriteOptions", inherit = ArrowObject)
CsvWriteOptions$create <- function(include_header = TRUE, batch_size = 1024L){
assert_that(is_integerish(batch_size, n = 1, finite = TRUE), batch_size > 0)
csv___WriteOptions__initialize(
list(
include_header = include_header,
batch_size = as.integer(batch_size)
)
)
}

readr_to_csv_read_options <- function(skip, col_names, col_types) {
if (isTRUE(col_names)) {
# C++ default to parse is 0-length string array
Expand DownExpand Up@@ -585,3 +603,49 @@ readr_to_csv_convert_options <- function(na,
include_columns = include_columns
)
}

#' Write CSV file to disk
#'
#' @param x `data.frame`, [RecordBatch], or [Table]
#' @param sink A string file path, URI, or [OutputStream], or path in a file
#' system (`SubTreeFileSystem`)
#' @param include_header Whether to write an initial header line with column names
#' @param batch_size Maximum number of rows processed at a time. Default is 1024.
#'
#' @return The input `x`, invisibly. Note that if `sink` is an [OutputStream],
#' the stream will be left open.
#' @export
#' @examples
#' \donttest{
#' tf <- tempfile()
#' on.exit(unlink(tf))
#' write_csv_arrow(mtcars, tf)
#' }
#' @include arrow-package.R
write_csv_arrow <- function(x,
sink,
include_header = TRUE,
batch_size = 1024L) {

write_options <- CsvWriteOptions$create(include_header, batch_size)

x_out <- x
if (is.data.frame(x)) {
x <- Table$create(x)
}

assert_is(x, "ArrowTabular")

if (!inherits(sink, "OutputStream")) {
sink <- make_output_stream(sink)
on.exit(sink$close())
}

if(inherits(x, "RecordBatch")){
csv___WriteCSV__RecordBatch(x, write_options, sink)
} else if(inherits(x, "Table")){
csv___WriteCSV__Table(x, write_options, sink)
}

invisible(x_out)
}
2 changes: 2 additions & 0 deletions r/_pkgdown.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -98,6 +98,7 @@ reference:
- write_ipc_stream
- write_to_raw
- write_parquet
- write_csv_arrow
- title: C++ reader/writer interface
contents:
- ParquetFileReader
Expand All@@ -109,6 +110,7 @@ reference:
- RecordBatchReader
- RecordBatchWriter
- CsvReadOptions
- CsvWriteOptions
- title: Arrow data containers
contents:
- array
Expand Down
22 changes: 22 additions & 0 deletions r/man/CsvWriteOptions.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

32 changes: 32 additions & 0 deletions r/man/write_csv_arrow.Rd

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

54 changes: 54 additions & 0 deletions r/src/arrowExports.cpp

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions r/src/arrow_types.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -178,6 +178,7 @@ R6_CLASS_NAME(arrow::csv::ReadOptions, "CsvReadOptions");
R6_CLASS_NAME(arrow::csv::ParseOptions, "CsvParseOptions");
R6_CLASS_NAME(arrow::csv::ConvertOptions, "CsvConvertOptions");
R6_CLASS_NAME(arrow::csv::TableReader, "CsvTableReader");
R6_CLASS_NAME(arrow::csv::WriteOptions, "CsvWriteOptions");

#if defined(ARROW_R_WITH_PARQUET)
R6_CLASS_NAME(parquet::ArrowReaderProperties, "ParquetArrowReaderProperties");
Expand Down
30 changes: 30 additions & 0 deletions r/src/csv.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,8 +20,21 @@
#if defined(ARROW_R_WITH_ARROW)

#include <arrow/csv/reader.h>
#include <arrow/csv/writer.h>
#include <arrow/memory_pool.h>

#include <arrow/util/value_parsing.h>

// [[arrow::export]]
std::shared_ptr<arrow::csv::WriteOptions> csv___WriteOptions__initialize(
cpp11::list options) {
auto res =
std::make_shared<arrow::csv::WriteOptions>(arrow::csv::WriteOptions::Defaults());
res->include_header = cpp11::as_cpp<bool>(options["include_header"]);
res->batch_size = cpp11::as_cpp<int>(options["batch_size"]);
return res;
}

// [[arrow::export]]
std::shared_ptr<arrow::csv::ReadOptions> csv___ReadOptions__initialize(
cpp11::list options) {
Expand DownExpand Up@@ -174,4 +187,21 @@ std::shared_ptr<arrow::TimestampParser> TimestampParser__MakeISO8601() {
return arrow::TimestampParser::MakeISO8601();
}

// [[arrow::export]]
void csv___WriteCSV__Table(const std::shared_ptr<arrow::Table>& table,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(
arrow::csv::WriteCSV(*table, *write_options, gc_memory_pool(), stream.get()));
}

// [[arrow::export]]
void csv___WriteCSV__RecordBatch(
const std::shared_ptr<arrow::RecordBatch>& record_batch,
const std::shared_ptr<arrow::csv::WriteOptions>& write_options,
const std::shared_ptr<arrow::io::OutputStream>& stream) {
StopIfNotOk(arrow::csv::WriteCSV(*record_batch, *write_options, gc_memory_pool(),
stream.get()));
}

#endif
Loading