Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 4 additions & 10 deletions dev/archery/archery/integration/datagen.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1493,32 +1493,26 @@ def _temp_path():

file_objs = [
generate_primitive_case([], name='primitive_no_batches'),
generate_primitive_case([17, 20], name='primitive')
.skip_category('Rust'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength')
.skip_category('Rust'),
generate_primitive_case([17, 20], name='primitive'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength'),

generate_primitive_large_offsets_case([17, 20])
.skip_category('Go')
.skip_category('JS')
.skip_category('Rust'),
.skip_category('JS'),

generate_null_case([10, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_null_trivial_case([0, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_decimal_case()
.skip_category('Go') # TODO(ARROW-7948): Decimal + Go
.skip_category('Rust'),

generate_datetime_case()
.skip_category('Rust'),
generate_datetime_case(),

generate_interval_case()
.skip_category('JS') # TODO(ARROW-5239): Intervals + JS
Expand Down
15 changes: 9 additions & 6 deletions rust/arrow-flight/src/utils.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -29,33 +29,36 @@ use arrow::record_batch::RecordBatch;
/// Convert a `RecordBatch` to `FlightData` by getting the header and body as bytes
impl From<&RecordBatch> for FlightData {
fn from(batch: &RecordBatch) -> Self {
let (header, body) = writer::record_batch_to_bytes(batch);
let options = writer::IpcWriteOptions::default();
let data = writer::record_batch_to_bytes(batch, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: header,
data_body: body,
data_header: data.ipc_message,
data_body: data.arrow_data,
}
}
}

/// Convert a `Schema` to `SchemaResult` by converting to an IPC message
impl From<&Schema> for SchemaResult {
fn from(schema: &Schema) -> Self {
let options = writer::IpcWriteOptions::default();
Self {
schema: writer::schema_to_bytes(schema),
schema: writer::schema_to_bytes(schema, &options).ipc_message,
}
}
}

/// Convert a `Schema` to `FlightData` by converting to an IPC message
impl From<&Schema> for FlightData {
fn from(schema: &Schema) -> Self {
let schema = writer::schema_to_bytes(schema);
let options = writer::IpcWriteOptions::default();
let schema = writer::schema_to_bytes(schema, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: schema,
data_header: schema.ipc_message,
data_body: vec![],
}
}
Expand Down
2 changes: 1 addition & 1 deletion rust/arrow/src/ipc/convert.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -345,7 +345,7 @@ pub(crate) fn get_fb_field_type<'a: 'b, 'b>(
Null => FBFieldType {
type_type: ipc::Type::Null,
type_: ipc::NullBuilder::new(fbb).finish().as_union_value(),
children: None,
children: Some(fbb.create_vector(&empty_fields[..])),
},
Boolean => FBFieldType {
type_type: ipc::Type::Bool,
Expand Down
1 change: 1 addition & 0 deletions rust/arrow/src/ipc/mod.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -36,3 +36,4 @@ pub use self::gen::SparseTensor::*;
pub use self::gen::Tensor::*;

static ARROW_MAGIC: [u8; 6] = [b'A', b'R', b'R', b'O', b'W', b'1'];

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is const better than static here? IIRC I used static for ARROW_MAGIC when const either wasn't a thing yet, or I was unfamiliar with it.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yea, I think this can be const.

static CONTINUATION_MARKER: [u8; 4] = [0xff; 4];
77 changes: 51 additions & 26 deletions rust/arrow/src/ipc/reader.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,9 +31,9 @@ use crate::datatypes::{DataType, Field, IntervalUnit, Schema, SchemaRef};
use crate::error::{ArrowError, Result};
use crate::ipc;
use crate::record_batch::{RecordBatch, RecordBatchReader};
use DataType::*;

const CONTINUATION_MARKER: u32 = 0xffff_ffff;
use ipc::CONTINUATION_MARKER;
use DataType::*;

/// Read a buffer based on offset and length
fn read_buffer(buf: &ipc::Buffer, a_data: &[u8]) -> Buffer {
Expand DownExpand Up@@ -482,6 +482,9 @@ pub struct FileReader<R: Read + Seek> {
///
/// Dictionaries may be appended to in the streaming format.
dictionaries_by_field: Vec<Option<ArrayRef>>,

/// Metadata version
metadata_version: ipc::MetadataVersion,
}

impl<R: Read + Seek> FileReader<R> {
Expand All@@ -506,12 +509,11 @@ impl<R: Read + Seek> FileReader<R> {
"Arrow file does not contain correct footer".to_string(),
));
}

// what does the footer contain?
// read footer length
let mut footer_size: [u8; 4] = [0; 4];
reader.seek(SeekFrom::End(-10))?;
reader.read_exact(&mut footer_size)?;
let footer_len = u32::from_le_bytes(footer_size);
let footer_len = i32::from_le_bytes(footer_size);

// read footer
let mut footer_data = vec![0; footer_len as usize];
Expand All@@ -534,6 +536,7 @@ impl<R: Read + Seek> FileReader<R> {
let mut dictionaries_by_field = vec![None; schema.fields().len()];
for block in footer.dictionaries().unwrap() {
// read length from end of offset
// TODO: ARROW-9848: dictionary metadata has not been tested
let meta_len = block.metaDataLength() - 4;

let mut block_data = vec![0; meta_len as usize];
Expand All@@ -554,15 +557,21 @@ impl<R: Read + Seek> FileReader<R> {
reader.read_exact(&mut buf)?;

if batch.isDelta() {
panic!("delta dictionary batches not supported");
return Err(ArrowError::IoError(
"delta dictionary batches not supported".to_string(),
));
}

let id = batch.id();

// As the dictionary batch does not contain the type of the
// values array, we need to retieve this from the schema.
let first_field = find_dictionary_field(&ipc_schema, id)
.expect("dictionary id not found in shchema");
let first_field =
find_dictionary_field(&ipc_schema, id).ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// Get an array representing this dictionary's values.
let dictionary_values: ArrayRef =
Expand All@@ -589,7 +598,11 @@ impl<R: Read + Seek> FileReader<R> {
}
_ => None,
}
.expect("dictionary id not found in schema");
.ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// for all fields with this dictionary id, update the dictionaries vector
// in the reader. Note that a dictionary batch may be shared between many fields.
Expand All@@ -606,7 +619,11 @@ impl<R: Read + Seek> FileReader<R> {
}
}
}
_ => panic!("Expecting DictionaryBatch in dictionary blocks."),
_ => {
return Err(ArrowError::IoError(
"Expecting DictionaryBatch in dictionary blocks.".to_string(),
))
}
};
}

Expand All@@ -617,6 +634,7 @@ impl<R: Read + Seek> FileReader<R> {
current_block: 0,
total_blocks,
dictionaries_by_field,
metadata_version: footer.version(),
})
}

Expand DownExpand Up@@ -657,16 +675,31 @@ impl<R: Read + Seek> RecordBatchReader for FileReader<R> {
let block = self.blocks[self.current_block];
self.current_block += 1;

// read length from end of offset
let meta_len = block.metaDataLength() - 4;
// read length
self.reader.seek(SeekFrom::Start(block.offset() as u64))?;
let mut meta_buf = [0; 4];
self.reader.read_exact(&mut meta_buf)?;
if meta_buf == CONTINUATION_MARKER {
// continuation marker encountered, read message next
self.reader.read_exact(&mut meta_buf)?;
}
let meta_len = i32::from_le_bytes(meta_buf);

let mut block_data = vec![0; meta_len as usize];
self.reader
.seek(SeekFrom::Start(block.offset() as u64 + 4))?;
self.reader.read_exact(&mut block_data)?;

let message = ipc::get_root_as_message(&block_data[..]);

// some old test data's footer metadata is not set, so we account for that
if self.metadata_version != ipc::MetadataVersion::V1
&& message.version() != self.metadata_version
{
return Err(ArrowError::IoError(
"Could not read IPC message as metadata versions mismatch"
.to_string(),
));
}

match message.header_type() {
ipc::MessageHeader::Schema => Err(ArrowError::IoError(
"Not expecting a schema when messages are read".to_string(),
Expand DownExpand Up@@ -733,16 +766,12 @@ impl<R: Read> StreamReader<R> {
let mut meta_size: [u8; 4] = [0; 4];
reader.read_exact(&mut meta_size)?;
let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

let mut meta_buffer = vec![0; meta_len as usize];
Expand DownExpand Up@@ -806,16 +835,12 @@ impl<R: Read> RecordBatchReader for StreamReader<R> {
}

let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
self.reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

if meta_len == 0 {
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 4 additions & 10 deletions dev/archery/archery/integration/datagen.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1493,32 +1493,26 @@ def _temp_path():

file_objs = [
generate_primitive_case([], name='primitive_no_batches'),
generate_primitive_case([17, 20], name='primitive')
.skip_category('Rust'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength')
.skip_category('Rust'),
generate_primitive_case([17, 20], name='primitive'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength'),

generate_primitive_large_offsets_case([17, 20])
.skip_category('Go')
.skip_category('JS')
.skip_category('Rust'),
.skip_category('JS'),

generate_null_case([10, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_null_trivial_case([0, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_decimal_case()
.skip_category('Go') # TODO(ARROW-7948): Decimal + Go
.skip_category('Rust'),

generate_datetime_case()
.skip_category('Rust'),
generate_datetime_case(),

generate_interval_case()
.skip_category('JS') # TODO(ARROW-5239): Intervals + JS
Expand Down
15 changes: 9 additions & 6 deletions rust/arrow-flight/src/utils.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -29,33 +29,36 @@ use arrow::record_batch::RecordBatch;
/// Convert a `RecordBatch` to `FlightData` by getting the header and body as bytes
impl From<&RecordBatch> for FlightData {
fn from(batch: &RecordBatch) -> Self {
let (header, body) = writer::record_batch_to_bytes(batch);
let options = writer::IpcWriteOptions::default();
let data = writer::record_batch_to_bytes(batch, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: header,
data_body: body,
data_header: data.ipc_message,
data_body: data.arrow_data,
}
}
}

/// Convert a `Schema` to `SchemaResult` by converting to an IPC message
impl From<&Schema> for SchemaResult {
fn from(schema: &Schema) -> Self {
let options = writer::IpcWriteOptions::default();
Self {
schema: writer::schema_to_bytes(schema),
schema: writer::schema_to_bytes(schema, &options).ipc_message,
}
}
}

/// Convert a `Schema` to `FlightData` by converting to an IPC message
impl From<&Schema> for FlightData {
fn from(schema: &Schema) -> Self {
let schema = writer::schema_to_bytes(schema);
let options = writer::IpcWriteOptions::default();
let schema = writer::schema_to_bytes(schema, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: schema,
data_header: schema.ipc_message,
data_body: vec![],
}
}
Expand Down
2 changes: 1 addition & 1 deletion rust/arrow/src/ipc/convert.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -345,7 +345,7 @@ pub(crate) fn get_fb_field_type<'a: 'b, 'b>(
Null => FBFieldType {
type_type: ipc::Type::Null,
type_: ipc::NullBuilder::new(fbb).finish().as_union_value(),
children: None,
children: Some(fbb.create_vector(&empty_fields[..])),
},
Boolean => FBFieldType {
type_type: ipc::Type::Bool,
Expand Down
1 change: 1 addition & 0 deletions rust/arrow/src/ipc/mod.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -36,3 +36,4 @@ pub use self::gen::SparseTensor::*;
pub use self::gen::Tensor::*;

static ARROW_MAGIC: [u8; 6] = [b'A', b'R', b'R', b'O', b'W', b'1'];

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is const better than static here? IIRC I used static for ARROW_MAGIC when const either wasn't a thing yet, or I was unfamiliar with it.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yea, I think this can be const.

static CONTINUATION_MARKER: [u8; 4] = [0xff; 4];
77 changes: 51 additions & 26 deletions rust/arrow/src/ipc/reader.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,9 +31,9 @@ use crate::datatypes::{DataType, Field, IntervalUnit, Schema, SchemaRef};
use crate::error::{ArrowError, Result};
use crate::ipc;
use crate::record_batch::{RecordBatch, RecordBatchReader};
use DataType::*;

const CONTINUATION_MARKER: u32 = 0xffff_ffff;
use ipc::CONTINUATION_MARKER;
use DataType::*;

/// Read a buffer based on offset and length
fn read_buffer(buf: &ipc::Buffer, a_data: &[u8]) -> Buffer {
Expand DownExpand Up@@ -482,6 +482,9 @@ pub struct FileReader<R: Read + Seek> {
///
/// Dictionaries may be appended to in the streaming format.
dictionaries_by_field: Vec<Option<ArrayRef>>,

/// Metadata version
metadata_version: ipc::MetadataVersion,
}

impl<R: Read + Seek> FileReader<R> {
Expand All@@ -506,12 +509,11 @@ impl<R: Read + Seek> FileReader<R> {
"Arrow file does not contain correct footer".to_string(),
));
}

// what does the footer contain?
// read footer length
let mut footer_size: [u8; 4] = [0; 4];
reader.seek(SeekFrom::End(-10))?;
reader.read_exact(&mut footer_size)?;
let footer_len = u32::from_le_bytes(footer_size);
let footer_len = i32::from_le_bytes(footer_size);

// read footer
let mut footer_data = vec![0; footer_len as usize];
Expand All@@ -534,6 +536,7 @@ impl<R: Read + Seek> FileReader<R> {
let mut dictionaries_by_field = vec![None; schema.fields().len()];
for block in footer.dictionaries().unwrap() {
// read length from end of offset
// TODO: ARROW-9848: dictionary metadata has not been tested
let meta_len = block.metaDataLength() - 4;

let mut block_data = vec![0; meta_len as usize];
Expand All@@ -554,15 +557,21 @@ impl<R: Read + Seek> FileReader<R> {
reader.read_exact(&mut buf)?;

if batch.isDelta() {
panic!("delta dictionary batches not supported");
return Err(ArrowError::IoError(
"delta dictionary batches not supported".to_string(),
));
}

let id = batch.id();

// As the dictionary batch does not contain the type of the
// values array, we need to retieve this from the schema.
let first_field = find_dictionary_field(&ipc_schema, id)
.expect("dictionary id not found in shchema");
let first_field =
find_dictionary_field(&ipc_schema, id).ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// Get an array representing this dictionary's values.
let dictionary_values: ArrayRef =
Expand All@@ -589,7 +598,11 @@ impl<R: Read + Seek> FileReader<R> {
}
_ => None,
}
.expect("dictionary id not found in schema");
.ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// for all fields with this dictionary id, update the dictionaries vector
// in the reader. Note that a dictionary batch may be shared between many fields.
Expand All@@ -606,7 +619,11 @@ impl<R: Read + Seek> FileReader<R> {
}
}
}
_ => panic!("Expecting DictionaryBatch in dictionary blocks."),
_ => {
return Err(ArrowError::IoError(
"Expecting DictionaryBatch in dictionary blocks.".to_string(),
))
}
};
}

Expand All@@ -617,6 +634,7 @@ impl<R: Read + Seek> FileReader<R> {
current_block: 0,
total_blocks,
dictionaries_by_field,
metadata_version: footer.version(),
})
}

Expand DownExpand Up@@ -657,16 +675,31 @@ impl<R: Read + Seek> RecordBatchReader for FileReader<R> {
let block = self.blocks[self.current_block];
self.current_block += 1;

// read length from end of offset
let meta_len = block.metaDataLength() - 4;
// read length
self.reader.seek(SeekFrom::Start(block.offset() as u64))?;
let mut meta_buf = [0; 4];
self.reader.read_exact(&mut meta_buf)?;
if meta_buf == CONTINUATION_MARKER {
// continuation marker encountered, read message next
self.reader.read_exact(&mut meta_buf)?;
}
let meta_len = i32::from_le_bytes(meta_buf);

let mut block_data = vec![0; meta_len as usize];
self.reader
.seek(SeekFrom::Start(block.offset() as u64 + 4))?;
self.reader.read_exact(&mut block_data)?;

let message = ipc::get_root_as_message(&block_data[..]);

// some old test data's footer metadata is not set, so we account for that
if self.metadata_version != ipc::MetadataVersion::V1
&& message.version() != self.metadata_version
{
return Err(ArrowError::IoError(
"Could not read IPC message as metadata versions mismatch"
.to_string(),
));
}

match message.header_type() {
ipc::MessageHeader::Schema => Err(ArrowError::IoError(
"Not expecting a schema when messages are read".to_string(),
Expand DownExpand Up@@ -733,16 +766,12 @@ impl<R: Read> StreamReader<R> {
let mut meta_size: [u8; 4] = [0; 4];
reader.read_exact(&mut meta_size)?;
let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

let mut meta_buffer = vec![0; meta_len as usize];
Expand DownExpand Up@@ -806,16 +835,12 @@ impl<R: Read> RecordBatchReader for StreamReader<R> {
}

let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
self.reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

if meta_len == 0 {
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 4 additions & 10 deletions dev/archery/archery/integration/datagen.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1493,32 +1493,26 @@ def _temp_path():

file_objs = [
generate_primitive_case([], name='primitive_no_batches'),
generate_primitive_case([17, 20], name='primitive')
.skip_category('Rust'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength')
.skip_category('Rust'),
generate_primitive_case([17, 20], name='primitive'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength'),

generate_primitive_large_offsets_case([17, 20])
.skip_category('Go')
.skip_category('JS')
.skip_category('Rust'),
.skip_category('JS'),

generate_null_case([10, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_null_trivial_case([0, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_decimal_case()
.skip_category('Go') # TODO(ARROW-7948): Decimal + Go
.skip_category('Rust'),

generate_datetime_case()
.skip_category('Rust'),
generate_datetime_case(),

generate_interval_case()
.skip_category('JS') # TODO(ARROW-5239): Intervals + JS
Expand Down
15 changes: 9 additions & 6 deletions rust/arrow-flight/src/utils.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -29,33 +29,36 @@ use arrow::record_batch::RecordBatch;
/// Convert a `RecordBatch` to `FlightData` by getting the header and body as bytes
impl From<&RecordBatch> for FlightData {
fn from(batch: &RecordBatch) -> Self {
let (header, body) = writer::record_batch_to_bytes(batch);
let options = writer::IpcWriteOptions::default();
let data = writer::record_batch_to_bytes(batch, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: header,
data_body: body,
data_header: data.ipc_message,
data_body: data.arrow_data,
}
}
}

/// Convert a `Schema` to `SchemaResult` by converting to an IPC message
impl From<&Schema> for SchemaResult {
fn from(schema: &Schema) -> Self {
let options = writer::IpcWriteOptions::default();
Self {
schema: writer::schema_to_bytes(schema),
schema: writer::schema_to_bytes(schema, &options).ipc_message,
}
}
}

/// Convert a `Schema` to `FlightData` by converting to an IPC message
impl From<&Schema> for FlightData {
fn from(schema: &Schema) -> Self {
let schema = writer::schema_to_bytes(schema);
let options = writer::IpcWriteOptions::default();
let schema = writer::schema_to_bytes(schema, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: schema,
data_header: schema.ipc_message,
data_body: vec![],
}
}
Expand Down
2 changes: 1 addition & 1 deletion rust/arrow/src/ipc/convert.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -345,7 +345,7 @@ pub(crate) fn get_fb_field_type<'a: 'b, 'b>(
Null => FBFieldType {
type_type: ipc::Type::Null,
type_: ipc::NullBuilder::new(fbb).finish().as_union_value(),
children: None,
children: Some(fbb.create_vector(&empty_fields[..])),
},
Boolean => FBFieldType {
type_type: ipc::Type::Bool,
Expand Down
1 change: 1 addition & 0 deletions rust/arrow/src/ipc/mod.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -36,3 +36,4 @@ pub use self::gen::SparseTensor::*;
pub use self::gen::Tensor::*;

static ARROW_MAGIC: [u8; 6] = [b'A', b'R', b'R', b'O', b'W', b'1'];

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is const better than static here? IIRC I used static for ARROW_MAGIC when const either wasn't a thing yet, or I was unfamiliar with it.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yea, I think this can be const.

static CONTINUATION_MARKER: [u8; 4] = [0xff; 4];
77 changes: 51 additions & 26 deletions rust/arrow/src/ipc/reader.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,9 +31,9 @@ use crate::datatypes::{DataType, Field, IntervalUnit, Schema, SchemaRef};
use crate::error::{ArrowError, Result};
use crate::ipc;
use crate::record_batch::{RecordBatch, RecordBatchReader};
use DataType::*;

const CONTINUATION_MARKER: u32 = 0xffff_ffff;
use ipc::CONTINUATION_MARKER;
use DataType::*;

/// Read a buffer based on offset and length
fn read_buffer(buf: &ipc::Buffer, a_data: &[u8]) -> Buffer {
Expand DownExpand Up@@ -482,6 +482,9 @@ pub struct FileReader<R: Read + Seek> {
///
/// Dictionaries may be appended to in the streaming format.
dictionaries_by_field: Vec<Option<ArrayRef>>,

/// Metadata version
metadata_version: ipc::MetadataVersion,
}

impl<R: Read + Seek> FileReader<R> {
Expand All@@ -506,12 +509,11 @@ impl<R: Read + Seek> FileReader<R> {
"Arrow file does not contain correct footer".to_string(),
));
}

// what does the footer contain?
// read footer length
let mut footer_size: [u8; 4] = [0; 4];
reader.seek(SeekFrom::End(-10))?;
reader.read_exact(&mut footer_size)?;
let footer_len = u32::from_le_bytes(footer_size);
let footer_len = i32::from_le_bytes(footer_size);

// read footer
let mut footer_data = vec![0; footer_len as usize];
Expand All@@ -534,6 +536,7 @@ impl<R: Read + Seek> FileReader<R> {
let mut dictionaries_by_field = vec![None; schema.fields().len()];
for block in footer.dictionaries().unwrap() {
// read length from end of offset
// TODO: ARROW-9848: dictionary metadata has not been tested
let meta_len = block.metaDataLength() - 4;

let mut block_data = vec![0; meta_len as usize];
Expand All@@ -554,15 +557,21 @@ impl<R: Read + Seek> FileReader<R> {
reader.read_exact(&mut buf)?;

if batch.isDelta() {
panic!("delta dictionary batches not supported");
return Err(ArrowError::IoError(
"delta dictionary batches not supported".to_string(),
));
}

let id = batch.id();

// As the dictionary batch does not contain the type of the
// values array, we need to retieve this from the schema.
let first_field = find_dictionary_field(&ipc_schema, id)
.expect("dictionary id not found in shchema");
let first_field =
find_dictionary_field(&ipc_schema, id).ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// Get an array representing this dictionary's values.
let dictionary_values: ArrayRef =
Expand All@@ -589,7 +598,11 @@ impl<R: Read + Seek> FileReader<R> {
}
_ => None,
}
.expect("dictionary id not found in schema");
.ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// for all fields with this dictionary id, update the dictionaries vector
// in the reader. Note that a dictionary batch may be shared between many fields.
Expand All@@ -606,7 +619,11 @@ impl<R: Read + Seek> FileReader<R> {
}
}
}
_ => panic!("Expecting DictionaryBatch in dictionary blocks."),
_ => {
return Err(ArrowError::IoError(
"Expecting DictionaryBatch in dictionary blocks.".to_string(),
))
}
};
}

Expand All@@ -617,6 +634,7 @@ impl<R: Read + Seek> FileReader<R> {
current_block: 0,
total_blocks,
dictionaries_by_field,
metadata_version: footer.version(),
})
}

Expand DownExpand Up@@ -657,16 +675,31 @@ impl<R: Read + Seek> RecordBatchReader for FileReader<R> {
let block = self.blocks[self.current_block];
self.current_block += 1;

// read length from end of offset
let meta_len = block.metaDataLength() - 4;
// read length
self.reader.seek(SeekFrom::Start(block.offset() as u64))?;
let mut meta_buf = [0; 4];
self.reader.read_exact(&mut meta_buf)?;
if meta_buf == CONTINUATION_MARKER {
// continuation marker encountered, read message next
self.reader.read_exact(&mut meta_buf)?;
}
let meta_len = i32::from_le_bytes(meta_buf);

let mut block_data = vec![0; meta_len as usize];
self.reader
.seek(SeekFrom::Start(block.offset() as u64 + 4))?;
self.reader.read_exact(&mut block_data)?;

let message = ipc::get_root_as_message(&block_data[..]);

// some old test data's footer metadata is not set, so we account for that
if self.metadata_version != ipc::MetadataVersion::V1
&& message.version() != self.metadata_version
{
return Err(ArrowError::IoError(
"Could not read IPC message as metadata versions mismatch"
.to_string(),
));
}

match message.header_type() {
ipc::MessageHeader::Schema => Err(ArrowError::IoError(
"Not expecting a schema when messages are read".to_string(),
Expand DownExpand Up@@ -733,16 +766,12 @@ impl<R: Read> StreamReader<R> {
let mut meta_size: [u8; 4] = [0; 4];
reader.read_exact(&mut meta_size)?;
let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

let mut meta_buffer = vec![0; meta_len as usize];
Expand DownExpand Up@@ -806,16 +835,12 @@ impl<R: Read> RecordBatchReader for StreamReader<R> {
}

let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
self.reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

if meta_len == 0 {
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 4 additions & 10 deletions dev/archery/archery/integration/datagen.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1493,32 +1493,26 @@ def _temp_path():

file_objs = [
generate_primitive_case([], name='primitive_no_batches'),
generate_primitive_case([17, 20], name='primitive')
.skip_category('Rust'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength')
.skip_category('Rust'),
generate_primitive_case([17, 20], name='primitive'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength'),

generate_primitive_large_offsets_case([17, 20])
.skip_category('Go')
.skip_category('JS')
.skip_category('Rust'),
.skip_category('JS'),

generate_null_case([10, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_null_trivial_case([0, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_decimal_case()
.skip_category('Go') # TODO(ARROW-7948): Decimal + Go
.skip_category('Rust'),

generate_datetime_case()
.skip_category('Rust'),
generate_datetime_case(),

generate_interval_case()
.skip_category('JS') # TODO(ARROW-5239): Intervals + JS
Expand Down
15 changes: 9 additions & 6 deletions rust/arrow-flight/src/utils.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -29,33 +29,36 @@ use arrow::record_batch::RecordBatch;
/// Convert a `RecordBatch` to `FlightData` by getting the header and body as bytes
impl From<&RecordBatch> for FlightData {
fn from(batch: &RecordBatch) -> Self {
let (header, body) = writer::record_batch_to_bytes(batch);
let options = writer::IpcWriteOptions::default();
let data = writer::record_batch_to_bytes(batch, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: header,
data_body: body,
data_header: data.ipc_message,
data_body: data.arrow_data,
}
}
}

/// Convert a `Schema` to `SchemaResult` by converting to an IPC message
impl From<&Schema> for SchemaResult {
fn from(schema: &Schema) -> Self {
let options = writer::IpcWriteOptions::default();
Self {
schema: writer::schema_to_bytes(schema),
schema: writer::schema_to_bytes(schema, &options).ipc_message,
}
}
}

/// Convert a `Schema` to `FlightData` by converting to an IPC message
impl From<&Schema> for FlightData {
fn from(schema: &Schema) -> Self {
let schema = writer::schema_to_bytes(schema);
let options = writer::IpcWriteOptions::default();
let schema = writer::schema_to_bytes(schema, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: schema,
data_header: schema.ipc_message,
data_body: vec![],
}
}
Expand Down
2 changes: 1 addition & 1 deletion rust/arrow/src/ipc/convert.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -345,7 +345,7 @@ pub(crate) fn get_fb_field_type<'a: 'b, 'b>(
Null => FBFieldType {
type_type: ipc::Type::Null,
type_: ipc::NullBuilder::new(fbb).finish().as_union_value(),
children: None,
children: Some(fbb.create_vector(&empty_fields[..])),
},
Boolean => FBFieldType {
type_type: ipc::Type::Bool,
Expand Down
1 change: 1 addition & 0 deletions rust/arrow/src/ipc/mod.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -36,3 +36,4 @@ pub use self::gen::SparseTensor::*;
pub use self::gen::Tensor::*;

static ARROW_MAGIC: [u8; 6] = [b'A', b'R', b'R', b'O', b'W', b'1'];

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is const better than static here? IIRC I used static for ARROW_MAGIC when const either wasn't a thing yet, or I was unfamiliar with it.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yea, I think this can be const.

static CONTINUATION_MARKER: [u8; 4] = [0xff; 4];
77 changes: 51 additions & 26 deletions rust/arrow/src/ipc/reader.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,9 +31,9 @@ use crate::datatypes::{DataType, Field, IntervalUnit, Schema, SchemaRef};
use crate::error::{ArrowError, Result};
use crate::ipc;
use crate::record_batch::{RecordBatch, RecordBatchReader};
use DataType::*;

const CONTINUATION_MARKER: u32 = 0xffff_ffff;
use ipc::CONTINUATION_MARKER;
use DataType::*;

/// Read a buffer based on offset and length
fn read_buffer(buf: &ipc::Buffer, a_data: &[u8]) -> Buffer {
Expand DownExpand Up@@ -482,6 +482,9 @@ pub struct FileReader<R: Read + Seek> {
///
/// Dictionaries may be appended to in the streaming format.
dictionaries_by_field: Vec<Option<ArrayRef>>,

/// Metadata version
metadata_version: ipc::MetadataVersion,
}

impl<R: Read + Seek> FileReader<R> {
Expand All@@ -506,12 +509,11 @@ impl<R: Read + Seek> FileReader<R> {
"Arrow file does not contain correct footer".to_string(),
));
}

// what does the footer contain?
// read footer length
let mut footer_size: [u8; 4] = [0; 4];
reader.seek(SeekFrom::End(-10))?;
reader.read_exact(&mut footer_size)?;
let footer_len = u32::from_le_bytes(footer_size);
let footer_len = i32::from_le_bytes(footer_size);

// read footer
let mut footer_data = vec![0; footer_len as usize];
Expand All@@ -534,6 +536,7 @@ impl<R: Read + Seek> FileReader<R> {
let mut dictionaries_by_field = vec![None; schema.fields().len()];
for block in footer.dictionaries().unwrap() {
// read length from end of offset
// TODO: ARROW-9848: dictionary metadata has not been tested
let meta_len = block.metaDataLength() - 4;

let mut block_data = vec![0; meta_len as usize];
Expand All@@ -554,15 +557,21 @@ impl<R: Read + Seek> FileReader<R> {
reader.read_exact(&mut buf)?;

if batch.isDelta() {
panic!("delta dictionary batches not supported");
return Err(ArrowError::IoError(
"delta dictionary batches not supported".to_string(),
));
}

let id = batch.id();

// As the dictionary batch does not contain the type of the
// values array, we need to retieve this from the schema.
let first_field = find_dictionary_field(&ipc_schema, id)
.expect("dictionary id not found in shchema");
let first_field =
find_dictionary_field(&ipc_schema, id).ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// Get an array representing this dictionary's values.
let dictionary_values: ArrayRef =
Expand All@@ -589,7 +598,11 @@ impl<R: Read + Seek> FileReader<R> {
}
_ => None,
}
.expect("dictionary id not found in schema");
.ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// for all fields with this dictionary id, update the dictionaries vector
// in the reader. Note that a dictionary batch may be shared between many fields.
Expand All@@ -606,7 +619,11 @@ impl<R: Read + Seek> FileReader<R> {
}
}
}
_ => panic!("Expecting DictionaryBatch in dictionary blocks."),
_ => {
return Err(ArrowError::IoError(
"Expecting DictionaryBatch in dictionary blocks.".to_string(),
))
}
};
}

Expand All@@ -617,6 +634,7 @@ impl<R: Read + Seek> FileReader<R> {
current_block: 0,
total_blocks,
dictionaries_by_field,
metadata_version: footer.version(),
})
}

Expand DownExpand Up@@ -657,16 +675,31 @@ impl<R: Read + Seek> RecordBatchReader for FileReader<R> {
let block = self.blocks[self.current_block];
self.current_block += 1;

// read length from end of offset
let meta_len = block.metaDataLength() - 4;
// read length
self.reader.seek(SeekFrom::Start(block.offset() as u64))?;
let mut meta_buf = [0; 4];
self.reader.read_exact(&mut meta_buf)?;
if meta_buf == CONTINUATION_MARKER {
// continuation marker encountered, read message next
self.reader.read_exact(&mut meta_buf)?;
}
let meta_len = i32::from_le_bytes(meta_buf);

let mut block_data = vec![0; meta_len as usize];
self.reader
.seek(SeekFrom::Start(block.offset() as u64 + 4))?;
self.reader.read_exact(&mut block_data)?;

let message = ipc::get_root_as_message(&block_data[..]);

// some old test data's footer metadata is not set, so we account for that
if self.metadata_version != ipc::MetadataVersion::V1
&& message.version() != self.metadata_version
{
return Err(ArrowError::IoError(
"Could not read IPC message as metadata versions mismatch"
.to_string(),
));
}

match message.header_type() {
ipc::MessageHeader::Schema => Err(ArrowError::IoError(
"Not expecting a schema when messages are read".to_string(),
Expand DownExpand Up@@ -733,16 +766,12 @@ impl<R: Read> StreamReader<R> {
let mut meta_size: [u8; 4] = [0; 4];
reader.read_exact(&mut meta_size)?;
let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

let mut meta_buffer = vec![0; meta_len as usize];
Expand DownExpand Up@@ -806,16 +835,12 @@ impl<R: Read> RecordBatchReader for StreamReader<R> {
}

let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
self.reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

if meta_len == 0 {
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 4 additions & 10 deletions dev/archery/archery/integration/datagen.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1493,32 +1493,26 @@ def _temp_path():

file_objs = [
generate_primitive_case([], name='primitive_no_batches'),
generate_primitive_case([17, 20], name='primitive')
.skip_category('Rust'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength')
.skip_category('Rust'),
generate_primitive_case([17, 20], name='primitive'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength'),

generate_primitive_large_offsets_case([17, 20])
.skip_category('Go')
.skip_category('JS')
.skip_category('Rust'),
.skip_category('JS'),

generate_null_case([10, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_null_trivial_case([0, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_decimal_case()
.skip_category('Go') # TODO(ARROW-7948): Decimal + Go
.skip_category('Rust'),

generate_datetime_case()
.skip_category('Rust'),
generate_datetime_case(),

generate_interval_case()
.skip_category('JS') # TODO(ARROW-5239): Intervals + JS
Expand Down
15 changes: 9 additions & 6 deletions rust/arrow-flight/src/utils.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -29,33 +29,36 @@ use arrow::record_batch::RecordBatch;
/// Convert a `RecordBatch` to `FlightData` by getting the header and body as bytes
impl From<&RecordBatch> for FlightData {
fn from(batch: &RecordBatch) -> Self {
let (header, body) = writer::record_batch_to_bytes(batch);
let options = writer::IpcWriteOptions::default();
let data = writer::record_batch_to_bytes(batch, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: header,
data_body: body,
data_header: data.ipc_message,
data_body: data.arrow_data,
}
}
}

/// Convert a `Schema` to `SchemaResult` by converting to an IPC message
impl From<&Schema> for SchemaResult {
fn from(schema: &Schema) -> Self {
let options = writer::IpcWriteOptions::default();
Self {
schema: writer::schema_to_bytes(schema),
schema: writer::schema_to_bytes(schema, &options).ipc_message,
}
}
}

/// Convert a `Schema` to `FlightData` by converting to an IPC message
impl From<&Schema> for FlightData {
fn from(schema: &Schema) -> Self {
let schema = writer::schema_to_bytes(schema);
let options = writer::IpcWriteOptions::default();
let schema = writer::schema_to_bytes(schema, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: schema,
data_header: schema.ipc_message,
data_body: vec![],
}
}
Expand Down
2 changes: 1 addition & 1 deletion rust/arrow/src/ipc/convert.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -345,7 +345,7 @@ pub(crate) fn get_fb_field_type<'a: 'b, 'b>(
Null => FBFieldType {
type_type: ipc::Type::Null,
type_: ipc::NullBuilder::new(fbb).finish().as_union_value(),
children: None,
children: Some(fbb.create_vector(&empty_fields[..])),
},
Boolean => FBFieldType {
type_type: ipc::Type::Bool,
Expand Down
1 change: 1 addition & 0 deletions rust/arrow/src/ipc/mod.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -36,3 +36,4 @@ pub use self::gen::SparseTensor::*;
pub use self::gen::Tensor::*;

static ARROW_MAGIC: [u8; 6] = [b'A', b'R', b'R', b'O', b'W', b'1'];

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is const better than static here? IIRC I used static for ARROW_MAGIC when const either wasn't a thing yet, or I was unfamiliar with it.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yea, I think this can be const.

static CONTINUATION_MARKER: [u8; 4] = [0xff; 4];
77 changes: 51 additions & 26 deletions rust/arrow/src/ipc/reader.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,9 +31,9 @@ use crate::datatypes::{DataType, Field, IntervalUnit, Schema, SchemaRef};
use crate::error::{ArrowError, Result};
use crate::ipc;
use crate::record_batch::{RecordBatch, RecordBatchReader};
use DataType::*;

const CONTINUATION_MARKER: u32 = 0xffff_ffff;
use ipc::CONTINUATION_MARKER;
use DataType::*;

/// Read a buffer based on offset and length
fn read_buffer(buf: &ipc::Buffer, a_data: &[u8]) -> Buffer {
Expand DownExpand Up@@ -482,6 +482,9 @@ pub struct FileReader<R: Read + Seek> {
///
/// Dictionaries may be appended to in the streaming format.
dictionaries_by_field: Vec<Option<ArrayRef>>,

/// Metadata version
metadata_version: ipc::MetadataVersion,
}

impl<R: Read + Seek> FileReader<R> {
Expand All@@ -506,12 +509,11 @@ impl<R: Read + Seek> FileReader<R> {
"Arrow file does not contain correct footer".to_string(),
));
}

// what does the footer contain?
// read footer length
let mut footer_size: [u8; 4] = [0; 4];
reader.seek(SeekFrom::End(-10))?;
reader.read_exact(&mut footer_size)?;
let footer_len = u32::from_le_bytes(footer_size);
let footer_len = i32::from_le_bytes(footer_size);

// read footer
let mut footer_data = vec![0; footer_len as usize];
Expand All@@ -534,6 +536,7 @@ impl<R: Read + Seek> FileReader<R> {
let mut dictionaries_by_field = vec![None; schema.fields().len()];
for block in footer.dictionaries().unwrap() {
// read length from end of offset
// TODO: ARROW-9848: dictionary metadata has not been tested
let meta_len = block.metaDataLength() - 4;

let mut block_data = vec![0; meta_len as usize];
Expand All@@ -554,15 +557,21 @@ impl<R: Read + Seek> FileReader<R> {
reader.read_exact(&mut buf)?;

if batch.isDelta() {
panic!("delta dictionary batches not supported");
return Err(ArrowError::IoError(
"delta dictionary batches not supported".to_string(),
));
}

let id = batch.id();

// As the dictionary batch does not contain the type of the
// values array, we need to retieve this from the schema.
let first_field = find_dictionary_field(&ipc_schema, id)
.expect("dictionary id not found in shchema");
let first_field =
find_dictionary_field(&ipc_schema, id).ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// Get an array representing this dictionary's values.
let dictionary_values: ArrayRef =
Expand All@@ -589,7 +598,11 @@ impl<R: Read + Seek> FileReader<R> {
}
_ => None,
}
.expect("dictionary id not found in schema");
.ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// for all fields with this dictionary id, update the dictionaries vector
// in the reader. Note that a dictionary batch may be shared between many fields.
Expand All@@ -606,7 +619,11 @@ impl<R: Read + Seek> FileReader<R> {
}
}
}
_ => panic!("Expecting DictionaryBatch in dictionary blocks."),
_ => {
return Err(ArrowError::IoError(
"Expecting DictionaryBatch in dictionary blocks.".to_string(),
))
}
};
}

Expand All@@ -617,6 +634,7 @@ impl<R: Read + Seek> FileReader<R> {
current_block: 0,
total_blocks,
dictionaries_by_field,
metadata_version: footer.version(),
})
}

Expand DownExpand Up@@ -657,16 +675,31 @@ impl<R: Read + Seek> RecordBatchReader for FileReader<R> {
let block = self.blocks[self.current_block];
self.current_block += 1;

// read length from end of offset
let meta_len = block.metaDataLength() - 4;
// read length
self.reader.seek(SeekFrom::Start(block.offset() as u64))?;
let mut meta_buf = [0; 4];
self.reader.read_exact(&mut meta_buf)?;
if meta_buf == CONTINUATION_MARKER {
// continuation marker encountered, read message next
self.reader.read_exact(&mut meta_buf)?;
}
let meta_len = i32::from_le_bytes(meta_buf);

let mut block_data = vec![0; meta_len as usize];
self.reader
.seek(SeekFrom::Start(block.offset() as u64 + 4))?;
self.reader.read_exact(&mut block_data)?;

let message = ipc::get_root_as_message(&block_data[..]);

// some old test data's footer metadata is not set, so we account for that
if self.metadata_version != ipc::MetadataVersion::V1
&& message.version() != self.metadata_version
{
return Err(ArrowError::IoError(
"Could not read IPC message as metadata versions mismatch"
.to_string(),
));
}

match message.header_type() {
ipc::MessageHeader::Schema => Err(ArrowError::IoError(
"Not expecting a schema when messages are read".to_string(),
Expand DownExpand Up@@ -733,16 +766,12 @@ impl<R: Read> StreamReader<R> {
let mut meta_size: [u8; 4] = [0; 4];
reader.read_exact(&mut meta_size)?;
let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

let mut meta_buffer = vec![0; meta_len as usize];
Expand DownExpand Up@@ -806,16 +835,12 @@ impl<R: Read> RecordBatchReader for StreamReader<R> {
}

let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
self.reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

if meta_len == 0 {
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 4 additions & 10 deletions dev/archery/archery/integration/datagen.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1493,32 +1493,26 @@ def _temp_path():

file_objs = [
generate_primitive_case([], name='primitive_no_batches'),
generate_primitive_case([17, 20], name='primitive')
.skip_category('Rust'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength')
.skip_category('Rust'),
generate_primitive_case([17, 20], name='primitive'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength'),

generate_primitive_large_offsets_case([17, 20])
.skip_category('Go')
.skip_category('JS')
.skip_category('Rust'),
.skip_category('JS'),

generate_null_case([10, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_null_trivial_case([0, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_decimal_case()
.skip_category('Go') # TODO(ARROW-7948): Decimal + Go
.skip_category('Rust'),

generate_datetime_case()
.skip_category('Rust'),
generate_datetime_case(),

generate_interval_case()
.skip_category('JS') # TODO(ARROW-5239): Intervals + JS
Expand Down
15 changes: 9 additions & 6 deletions rust/arrow-flight/src/utils.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -29,33 +29,36 @@ use arrow::record_batch::RecordBatch;
/// Convert a `RecordBatch` to `FlightData` by getting the header and body as bytes
impl From<&RecordBatch> for FlightData {
fn from(batch: &RecordBatch) -> Self {
let (header, body) = writer::record_batch_to_bytes(batch);
let options = writer::IpcWriteOptions::default();
let data = writer::record_batch_to_bytes(batch, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: header,
data_body: body,
data_header: data.ipc_message,
data_body: data.arrow_data,
}
}
}

/// Convert a `Schema` to `SchemaResult` by converting to an IPC message
impl From<&Schema> for SchemaResult {
fn from(schema: &Schema) -> Self {
let options = writer::IpcWriteOptions::default();
Self {
schema: writer::schema_to_bytes(schema),
schema: writer::schema_to_bytes(schema, &options).ipc_message,
}
}
}

/// Convert a `Schema` to `FlightData` by converting to an IPC message
impl From<&Schema> for FlightData {
fn from(schema: &Schema) -> Self {
let schema = writer::schema_to_bytes(schema);
let options = writer::IpcWriteOptions::default();
let schema = writer::schema_to_bytes(schema, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: schema,
data_header: schema.ipc_message,
data_body: vec![],
}
}
Expand Down
2 changes: 1 addition & 1 deletion rust/arrow/src/ipc/convert.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -345,7 +345,7 @@ pub(crate) fn get_fb_field_type<'a: 'b, 'b>(
Null => FBFieldType {
type_type: ipc::Type::Null,
type_: ipc::NullBuilder::new(fbb).finish().as_union_value(),
children: None,
children: Some(fbb.create_vector(&empty_fields[..])),
},
Boolean => FBFieldType {
type_type: ipc::Type::Bool,
Expand Down
1 change: 1 addition & 0 deletions rust/arrow/src/ipc/mod.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -36,3 +36,4 @@ pub use self::gen::SparseTensor::*;
pub use self::gen::Tensor::*;

static ARROW_MAGIC: [u8; 6] = [b'A', b'R', b'R', b'O', b'W', b'1'];

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is const better than static here? IIRC I used static for ARROW_MAGIC when const either wasn't a thing yet, or I was unfamiliar with it.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yea, I think this can be const.

static CONTINUATION_MARKER: [u8; 4] = [0xff; 4];
77 changes: 51 additions & 26 deletions rust/arrow/src/ipc/reader.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,9 +31,9 @@ use crate::datatypes::{DataType, Field, IntervalUnit, Schema, SchemaRef};
use crate::error::{ArrowError, Result};
use crate::ipc;
use crate::record_batch::{RecordBatch, RecordBatchReader};
use DataType::*;

const CONTINUATION_MARKER: u32 = 0xffff_ffff;
use ipc::CONTINUATION_MARKER;
use DataType::*;

/// Read a buffer based on offset and length
fn read_buffer(buf: &ipc::Buffer, a_data: &[u8]) -> Buffer {
Expand DownExpand Up@@ -482,6 +482,9 @@ pub struct FileReader<R: Read + Seek> {
///
/// Dictionaries may be appended to in the streaming format.
dictionaries_by_field: Vec<Option<ArrayRef>>,

/// Metadata version
metadata_version: ipc::MetadataVersion,
}

impl<R: Read + Seek> FileReader<R> {
Expand All@@ -506,12 +509,11 @@ impl<R: Read + Seek> FileReader<R> {
"Arrow file does not contain correct footer".to_string(),
));
}

// what does the footer contain?
// read footer length
let mut footer_size: [u8; 4] = [0; 4];
reader.seek(SeekFrom::End(-10))?;
reader.read_exact(&mut footer_size)?;
let footer_len = u32::from_le_bytes(footer_size);
let footer_len = i32::from_le_bytes(footer_size);

// read footer
let mut footer_data = vec![0; footer_len as usize];
Expand All@@ -534,6 +536,7 @@ impl<R: Read + Seek> FileReader<R> {
let mut dictionaries_by_field = vec![None; schema.fields().len()];
for block in footer.dictionaries().unwrap() {
// read length from end of offset
// TODO: ARROW-9848: dictionary metadata has not been tested
let meta_len = block.metaDataLength() - 4;

let mut block_data = vec![0; meta_len as usize];
Expand All@@ -554,15 +557,21 @@ impl<R: Read + Seek> FileReader<R> {
reader.read_exact(&mut buf)?;

if batch.isDelta() {
panic!("delta dictionary batches not supported");
return Err(ArrowError::IoError(
"delta dictionary batches not supported".to_string(),
));
}

let id = batch.id();

// As the dictionary batch does not contain the type of the
// values array, we need to retieve this from the schema.
let first_field = find_dictionary_field(&ipc_schema, id)
.expect("dictionary id not found in shchema");
let first_field =
find_dictionary_field(&ipc_schema, id).ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// Get an array representing this dictionary's values.
let dictionary_values: ArrayRef =
Expand All@@ -589,7 +598,11 @@ impl<R: Read + Seek> FileReader<R> {
}
_ => None,
}
.expect("dictionary id not found in schema");
.ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// for all fields with this dictionary id, update the dictionaries vector
// in the reader. Note that a dictionary batch may be shared between many fields.
Expand All@@ -606,7 +619,11 @@ impl<R: Read + Seek> FileReader<R> {
}
}
}
_ => panic!("Expecting DictionaryBatch in dictionary blocks."),
_ => {
return Err(ArrowError::IoError(
"Expecting DictionaryBatch in dictionary blocks.".to_string(),
))
}
};
}

Expand All@@ -617,6 +634,7 @@ impl<R: Read + Seek> FileReader<R> {
current_block: 0,
total_blocks,
dictionaries_by_field,
metadata_version: footer.version(),
})
}

Expand DownExpand Up@@ -657,16 +675,31 @@ impl<R: Read + Seek> RecordBatchReader for FileReader<R> {
let block = self.blocks[self.current_block];
self.current_block += 1;

// read length from end of offset
let meta_len = block.metaDataLength() - 4;
// read length
self.reader.seek(SeekFrom::Start(block.offset() as u64))?;
let mut meta_buf = [0; 4];
self.reader.read_exact(&mut meta_buf)?;
if meta_buf == CONTINUATION_MARKER {
// continuation marker encountered, read message next
self.reader.read_exact(&mut meta_buf)?;
}
let meta_len = i32::from_le_bytes(meta_buf);

let mut block_data = vec![0; meta_len as usize];
self.reader
.seek(SeekFrom::Start(block.offset() as u64 + 4))?;
self.reader.read_exact(&mut block_data)?;

let message = ipc::get_root_as_message(&block_data[..]);

// some old test data's footer metadata is not set, so we account for that
if self.metadata_version != ipc::MetadataVersion::V1
&& message.version() != self.metadata_version
{
return Err(ArrowError::IoError(
"Could not read IPC message as metadata versions mismatch"
.to_string(),
));
}

match message.header_type() {
ipc::MessageHeader::Schema => Err(ArrowError::IoError(
"Not expecting a schema when messages are read".to_string(),
Expand DownExpand Up@@ -733,16 +766,12 @@ impl<R: Read> StreamReader<R> {
let mut meta_size: [u8; 4] = [0; 4];
reader.read_exact(&mut meta_size)?;
let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

let mut meta_buffer = vec![0; meta_len as usize];
Expand DownExpand Up@@ -806,16 +835,12 @@ impl<R: Read> RecordBatchReader for StreamReader<R> {
}

let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
self.reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

if meta_len == 0 {
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 4 additions & 10 deletions dev/archery/archery/integration/datagen.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1493,32 +1493,26 @@ def _temp_path():

file_objs = [
generate_primitive_case([], name='primitive_no_batches'),
generate_primitive_case([17, 20], name='primitive')
.skip_category('Rust'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength')
.skip_category('Rust'),
generate_primitive_case([17, 20], name='primitive'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength'),

generate_primitive_large_offsets_case([17, 20])
.skip_category('Go')
.skip_category('JS')
.skip_category('Rust'),
.skip_category('JS'),

generate_null_case([10, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_null_trivial_case([0, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_decimal_case()
.skip_category('Go') # TODO(ARROW-7948): Decimal + Go
.skip_category('Rust'),

generate_datetime_case()
.skip_category('Rust'),
generate_datetime_case(),

generate_interval_case()
.skip_category('JS') # TODO(ARROW-5239): Intervals + JS
Expand Down
15 changes: 9 additions & 6 deletions rust/arrow-flight/src/utils.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -29,33 +29,36 @@ use arrow::record_batch::RecordBatch;
/// Convert a `RecordBatch` to `FlightData` by getting the header and body as bytes
impl From<&RecordBatch> for FlightData {
fn from(batch: &RecordBatch) -> Self {
let (header, body) = writer::record_batch_to_bytes(batch);
let options = writer::IpcWriteOptions::default();
let data = writer::record_batch_to_bytes(batch, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: header,
data_body: body,
data_header: data.ipc_message,
data_body: data.arrow_data,
}
}
}

/// Convert a `Schema` to `SchemaResult` by converting to an IPC message
impl From<&Schema> for SchemaResult {
fn from(schema: &Schema) -> Self {
let options = writer::IpcWriteOptions::default();
Self {
schema: writer::schema_to_bytes(schema),
schema: writer::schema_to_bytes(schema, &options).ipc_message,
}
}
}

/// Convert a `Schema` to `FlightData` by converting to an IPC message
impl From<&Schema> for FlightData {
fn from(schema: &Schema) -> Self {
let schema = writer::schema_to_bytes(schema);
let options = writer::IpcWriteOptions::default();
let schema = writer::schema_to_bytes(schema, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: schema,
data_header: schema.ipc_message,
data_body: vec![],
}
}
Expand Down
2 changes: 1 addition & 1 deletion rust/arrow/src/ipc/convert.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -345,7 +345,7 @@ pub(crate) fn get_fb_field_type<'a: 'b, 'b>(
Null => FBFieldType {
type_type: ipc::Type::Null,
type_: ipc::NullBuilder::new(fbb).finish().as_union_value(),
children: None,
children: Some(fbb.create_vector(&empty_fields[..])),
},
Boolean => FBFieldType {
type_type: ipc::Type::Bool,
Expand Down
1 change: 1 addition & 0 deletions rust/arrow/src/ipc/mod.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -36,3 +36,4 @@ pub use self::gen::SparseTensor::*;
pub use self::gen::Tensor::*;

static ARROW_MAGIC: [u8; 6] = [b'A', b'R', b'R', b'O', b'W', b'1'];

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is const better than static here? IIRC I used static for ARROW_MAGIC when const either wasn't a thing yet, or I was unfamiliar with it.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yea, I think this can be const.

static CONTINUATION_MARKER: [u8; 4] = [0xff; 4];
77 changes: 51 additions & 26 deletions rust/arrow/src/ipc/reader.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,9 +31,9 @@ use crate::datatypes::{DataType, Field, IntervalUnit, Schema, SchemaRef};
use crate::error::{ArrowError, Result};
use crate::ipc;
use crate::record_batch::{RecordBatch, RecordBatchReader};
use DataType::*;

const CONTINUATION_MARKER: u32 = 0xffff_ffff;
use ipc::CONTINUATION_MARKER;
use DataType::*;

/// Read a buffer based on offset and length
fn read_buffer(buf: &ipc::Buffer, a_data: &[u8]) -> Buffer {
Expand DownExpand Up@@ -482,6 +482,9 @@ pub struct FileReader<R: Read + Seek> {
///
/// Dictionaries may be appended to in the streaming format.
dictionaries_by_field: Vec<Option<ArrayRef>>,

/// Metadata version
metadata_version: ipc::MetadataVersion,
}

impl<R: Read + Seek> FileReader<R> {
Expand All@@ -506,12 +509,11 @@ impl<R: Read + Seek> FileReader<R> {
"Arrow file does not contain correct footer".to_string(),
));
}

// what does the footer contain?
// read footer length
let mut footer_size: [u8; 4] = [0; 4];
reader.seek(SeekFrom::End(-10))?;
reader.read_exact(&mut footer_size)?;
let footer_len = u32::from_le_bytes(footer_size);
let footer_len = i32::from_le_bytes(footer_size);

// read footer
let mut footer_data = vec![0; footer_len as usize];
Expand All@@ -534,6 +536,7 @@ impl<R: Read + Seek> FileReader<R> {
let mut dictionaries_by_field = vec![None; schema.fields().len()];
for block in footer.dictionaries().unwrap() {
// read length from end of offset
// TODO: ARROW-9848: dictionary metadata has not been tested
let meta_len = block.metaDataLength() - 4;

let mut block_data = vec![0; meta_len as usize];
Expand All@@ -554,15 +557,21 @@ impl<R: Read + Seek> FileReader<R> {
reader.read_exact(&mut buf)?;

if batch.isDelta() {
panic!("delta dictionary batches not supported");
return Err(ArrowError::IoError(
"delta dictionary batches not supported".to_string(),
));
}

let id = batch.id();

// As the dictionary batch does not contain the type of the
// values array, we need to retieve this from the schema.
let first_field = find_dictionary_field(&ipc_schema, id)
.expect("dictionary id not found in shchema");
let first_field =
find_dictionary_field(&ipc_schema, id).ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// Get an array representing this dictionary's values.
let dictionary_values: ArrayRef =
Expand All@@ -589,7 +598,11 @@ impl<R: Read + Seek> FileReader<R> {
}
_ => None,
}
.expect("dictionary id not found in schema");
.ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// for all fields with this dictionary id, update the dictionaries vector
// in the reader. Note that a dictionary batch may be shared between many fields.
Expand All@@ -606,7 +619,11 @@ impl<R: Read + Seek> FileReader<R> {
}
}
}
_ => panic!("Expecting DictionaryBatch in dictionary blocks."),
_ => {
return Err(ArrowError::IoError(
"Expecting DictionaryBatch in dictionary blocks.".to_string(),
))
}
};
}

Expand All@@ -617,6 +634,7 @@ impl<R: Read + Seek> FileReader<R> {
current_block: 0,
total_blocks,
dictionaries_by_field,
metadata_version: footer.version(),
})
}

Expand DownExpand Up@@ -657,16 +675,31 @@ impl<R: Read + Seek> RecordBatchReader for FileReader<R> {
let block = self.blocks[self.current_block];
self.current_block += 1;

// read length from end of offset
let meta_len = block.metaDataLength() - 4;
// read length
self.reader.seek(SeekFrom::Start(block.offset() as u64))?;
let mut meta_buf = [0; 4];
self.reader.read_exact(&mut meta_buf)?;
if meta_buf == CONTINUATION_MARKER {
// continuation marker encountered, read message next
self.reader.read_exact(&mut meta_buf)?;
}
let meta_len = i32::from_le_bytes(meta_buf);

let mut block_data = vec![0; meta_len as usize];
self.reader
.seek(SeekFrom::Start(block.offset() as u64 + 4))?;
self.reader.read_exact(&mut block_data)?;

let message = ipc::get_root_as_message(&block_data[..]);

// some old test data's footer metadata is not set, so we account for that
if self.metadata_version != ipc::MetadataVersion::V1
&& message.version() != self.metadata_version
{
return Err(ArrowError::IoError(
"Could not read IPC message as metadata versions mismatch"
.to_string(),
));
}

match message.header_type() {
ipc::MessageHeader::Schema => Err(ArrowError::IoError(
"Not expecting a schema when messages are read".to_string(),
Expand DownExpand Up@@ -733,16 +766,12 @@ impl<R: Read> StreamReader<R> {
let mut meta_size: [u8; 4] = [0; 4];
reader.read_exact(&mut meta_size)?;
let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

let mut meta_buffer = vec![0; meta_len as usize];
Expand DownExpand Up@@ -806,16 +835,12 @@ impl<R: Read> RecordBatchReader for StreamReader<R> {
}

let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
self.reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

if meta_len == 0 {
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 4 additions & 10 deletions dev/archery/archery/integration/datagen.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1493,32 +1493,26 @@ def _temp_path():

file_objs = [
generate_primitive_case([], name='primitive_no_batches'),
generate_primitive_case([17, 20], name='primitive')
.skip_category('Rust'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength')
.skip_category('Rust'),
generate_primitive_case([17, 20], name='primitive'),
generate_primitive_case([0, 0, 0], name='primitive_zerolength'),

generate_primitive_large_offsets_case([17, 20])
.skip_category('Go')
.skip_category('JS')
.skip_category('Rust'),
.skip_category('JS'),

generate_null_case([10, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_null_trivial_case([0, 0])
.skip_category('Rust')
.skip_category('JS') # TODO(ARROW-7900)
.skip_category('Go'), # TODO(ARROW-7901)

generate_decimal_case()
.skip_category('Go') # TODO(ARROW-7948): Decimal + Go
.skip_category('Rust'),

generate_datetime_case()
.skip_category('Rust'),
generate_datetime_case(),

generate_interval_case()
.skip_category('JS') # TODO(ARROW-5239): Intervals + JS
Expand Down
15 changes: 9 additions & 6 deletions rust/arrow-flight/src/utils.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -29,33 +29,36 @@ use arrow::record_batch::RecordBatch;
/// Convert a `RecordBatch` to `FlightData` by getting the header and body as bytes
impl From<&RecordBatch> for FlightData {
fn from(batch: &RecordBatch) -> Self {
let (header, body) = writer::record_batch_to_bytes(batch);
let options = writer::IpcWriteOptions::default();
let data = writer::record_batch_to_bytes(batch, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: header,
data_body: body,
data_header: data.ipc_message,
data_body: data.arrow_data,
}
}
}

/// Convert a `Schema` to `SchemaResult` by converting to an IPC message
impl From<&Schema> for SchemaResult {
fn from(schema: &Schema) -> Self {
let options = writer::IpcWriteOptions::default();
Self {
schema: writer::schema_to_bytes(schema),
schema: writer::schema_to_bytes(schema, &options).ipc_message,
}
}
}

/// Convert a `Schema` to `FlightData` by converting to an IPC message
impl From<&Schema> for FlightData {
fn from(schema: &Schema) -> Self {
let schema = writer::schema_to_bytes(schema);
let options = writer::IpcWriteOptions::default();
let schema = writer::schema_to_bytes(schema, &options);
Self {
flight_descriptor: None,
app_metadata: vec![],
data_header: schema,
data_header: schema.ipc_message,
data_body: vec![],
}
}
Expand Down
2 changes: 1 addition & 1 deletion rust/arrow/src/ipc/convert.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -345,7 +345,7 @@ pub(crate) fn get_fb_field_type<'a: 'b, 'b>(
Null => FBFieldType {
type_type: ipc::Type::Null,
type_: ipc::NullBuilder::new(fbb).finish().as_union_value(),
children: None,
children: Some(fbb.create_vector(&empty_fields[..])),
},
Boolean => FBFieldType {
type_type: ipc::Type::Bool,
Expand Down
1 change: 1 addition & 0 deletions rust/arrow/src/ipc/mod.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -36,3 +36,4 @@ pub use self::gen::SparseTensor::*;
pub use self::gen::Tensor::*;

static ARROW_MAGIC: [u8; 6] = [b'A', b'R', b'R', b'O', b'W', b'1'];

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

is const better than static here? IIRC I used static for ARROW_MAGIC when const either wasn't a thing yet, or I was unfamiliar with it.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yea, I think this can be const.

static CONTINUATION_MARKER: [u8; 4] = [0xff; 4];
77 changes: 51 additions & 26 deletions rust/arrow/src/ipc/reader.rs
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,9 +31,9 @@ use crate::datatypes::{DataType, Field, IntervalUnit, Schema, SchemaRef};
use crate::error::{ArrowError, Result};
use crate::ipc;
use crate::record_batch::{RecordBatch, RecordBatchReader};
use DataType::*;

const CONTINUATION_MARKER: u32 = 0xffff_ffff;
use ipc::CONTINUATION_MARKER;
use DataType::*;

/// Read a buffer based on offset and length
fn read_buffer(buf: &ipc::Buffer, a_data: &[u8]) -> Buffer {
Expand DownExpand Up@@ -482,6 +482,9 @@ pub struct FileReader<R: Read + Seek> {
///
/// Dictionaries may be appended to in the streaming format.
dictionaries_by_field: Vec<Option<ArrayRef>>,

/// Metadata version
metadata_version: ipc::MetadataVersion,
}

impl<R: Read + Seek> FileReader<R> {
Expand All@@ -506,12 +509,11 @@ impl<R: Read + Seek> FileReader<R> {
"Arrow file does not contain correct footer".to_string(),
));
}

// what does the footer contain?
// read footer length
let mut footer_size: [u8; 4] = [0; 4];
reader.seek(SeekFrom::End(-10))?;
reader.read_exact(&mut footer_size)?;
let footer_len = u32::from_le_bytes(footer_size);
let footer_len = i32::from_le_bytes(footer_size);

// read footer
let mut footer_data = vec![0; footer_len as usize];
Expand All@@ -534,6 +536,7 @@ impl<R: Read + Seek> FileReader<R> {
let mut dictionaries_by_field = vec![None; schema.fields().len()];
for block in footer.dictionaries().unwrap() {
// read length from end of offset
// TODO: ARROW-9848: dictionary metadata has not been tested
let meta_len = block.metaDataLength() - 4;

let mut block_data = vec![0; meta_len as usize];
Expand All@@ -554,15 +557,21 @@ impl<R: Read + Seek> FileReader<R> {
reader.read_exact(&mut buf)?;

if batch.isDelta() {
panic!("delta dictionary batches not supported");
return Err(ArrowError::IoError(
"delta dictionary batches not supported".to_string(),
));
}

let id = batch.id();

// As the dictionary batch does not contain the type of the
// values array, we need to retieve this from the schema.
let first_field = find_dictionary_field(&ipc_schema, id)
.expect("dictionary id not found in shchema");
let first_field =
find_dictionary_field(&ipc_schema, id).ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// Get an array representing this dictionary's values.
let dictionary_values: ArrayRef =
Expand All@@ -589,7 +598,11 @@ impl<R: Read + Seek> FileReader<R> {
}
_ => None,
}
.expect("dictionary id not found in schema");
.ok_or_else(|| {
ArrowError::InvalidArgumentError(
"dictionary id not found in schema".to_string(),
)
})?;

// for all fields with this dictionary id, update the dictionaries vector
// in the reader. Note that a dictionary batch may be shared between many fields.
Expand All@@ -606,7 +619,11 @@ impl<R: Read + Seek> FileReader<R> {
}
}
}
_ => panic!("Expecting DictionaryBatch in dictionary blocks."),
_ => {
return Err(ArrowError::IoError(
"Expecting DictionaryBatch in dictionary blocks.".to_string(),
))
}
};
}

Expand All@@ -617,6 +634,7 @@ impl<R: Read + Seek> FileReader<R> {
current_block: 0,
total_blocks,
dictionaries_by_field,
metadata_version: footer.version(),
})
}

Expand DownExpand Up@@ -657,16 +675,31 @@ impl<R: Read + Seek> RecordBatchReader for FileReader<R> {
let block = self.blocks[self.current_block];
self.current_block += 1;

// read length from end of offset
let meta_len = block.metaDataLength() - 4;
// read length
self.reader.seek(SeekFrom::Start(block.offset() as u64))?;
let mut meta_buf = [0; 4];
self.reader.read_exact(&mut meta_buf)?;
if meta_buf == CONTINUATION_MARKER {
// continuation marker encountered, read message next
self.reader.read_exact(&mut meta_buf)?;
}
let meta_len = i32::from_le_bytes(meta_buf);

let mut block_data = vec![0; meta_len as usize];
self.reader
.seek(SeekFrom::Start(block.offset() as u64 + 4))?;
self.reader.read_exact(&mut block_data)?;

let message = ipc::get_root_as_message(&block_data[..]);

// some old test data's footer metadata is not set, so we account for that
if self.metadata_version != ipc::MetadataVersion::V1
&& message.version() != self.metadata_version
{
return Err(ArrowError::IoError(
"Could not read IPC message as metadata versions mismatch"
.to_string(),
));
}

match message.header_type() {
ipc::MessageHeader::Schema => Err(ArrowError::IoError(
"Not expecting a schema when messages are read".to_string(),
Expand DownExpand Up@@ -733,16 +766,12 @@ impl<R: Read> StreamReader<R> {
let mut meta_size: [u8; 4] = [0; 4];
reader.read_exact(&mut meta_size)?;
let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

let mut meta_buffer = vec![0; meta_len as usize];
Expand DownExpand Up@@ -806,16 +835,12 @@ impl<R: Read> RecordBatchReader for StreamReader<R> {
}

let meta_len = {
let meta_len = u32::from_le_bytes(meta_size);

// If a continuation marker is encountered, skip over it and read
// the size from the next four bytes.
if meta_len == CONTINUATION_MARKER {
if meta_size == CONTINUATION_MARKER {
self.reader.read_exact(&mut meta_size)?;
u32::from_le_bytes(meta_size)
} else {
meta_len
}
i32::from_le_bytes(meta_size)
};

if meta_len == 0 {
Expand Down
Loading