Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
145 changes: 110 additions & 35 deletions src/Microsoft.ML.Parquet/ParquetLoader.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -12,6 +12,7 @@
using Microsoft.ML.Runtime;
using Microsoft.ML.Runtime.CommandLine;
using Microsoft.ML.Runtime.Data;
using Microsoft.ML.Runtime.Data.IO;
using Microsoft.ML.Runtime.Internal.Utilities;
using Microsoft.ML.Runtime.Model;
using Parquet;
Expand DownExpand Up@@ -88,48 +89,29 @@ public sealed class Arguments
internal const string ShortName = "Parquet";
internal const string ModelSignature = "PARQELDR";

private const string SchemaCtxName = "Schema.idv";

private readonly IHost _host;
private readonly Stream _parquetStream;
private readonly ParquetOptions _parquetOptions;
private readonly int _columnChunkReadSize;
private readonly Column[] _columnsLoaded;
private readonly DataSet _schemaDataSet;
private const int _defaultColumnChunkReadSize = 1000000;

private bool _disposed;
private long? _rowCount;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: ModelSignature,
verWrittenCur: 0x00010001, // Initial
verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does this commented out code provide value? Can it be removed?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

While we generally prefer to avoid checking in commented in code, the version history of past versions is different, since we want to know why we had to bump the version number each time, and we like to have that first version in there. See e.g., the text loader model, which is the most extreme example I am aware of.

verWrittenCur: 0x00010002, // Add Schema to Model Context
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag

Arguments args = new Arguments
{
ColumnChunkReadSize = ctx.Reader.ReadInt32(),
TreatBigIntegersAsDates = ctx.Reader.ReadBoolean()
};
return host.Apply("Loading Model",
ch => new ParquetLoader(args, host, OpenStream(files)));
}

public ParquetLoader(IHostEnvironment env, Arguments args, IMultiStreamSource files)
: this(env, args, OpenStream(files))
{
Expand DownExpand Up@@ -165,6 +147,8 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
TreatBigIntegersAsDates = args.TreatBigIntegersAsDates
};

DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
Expand All@@ -173,36 +157,112 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
Count = 0,
Offset = 0
};
_schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnChunkReadSize = args.ColumnChunkReadSize;
InitColumns(ch, out _columnsLoaded);
_columnsLoaded = InitColumns(schemaDataSet);
Schema = CreateSchema(_host, _columnsLoaded);
}
}

private ParquetLoader(IHost host, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.AssertValue(host);
_host = host;
_host.AssertValue(ctx);
_host.AssertValue(files);

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader (0x00010002)

_columnChunkReadSize = ctx.Reader.ReadInt32();
bool treatBigIntegersAsDates = ctx.Reader.ReadBoolean();

if (ctx.Header.ModelVerWritten >= 0x00010002)
{
// Load the schema
byte[] buffer = null;
if (!ctx.TryLoadBinaryStream(SchemaCtxName, r => buffer = r.ReadByteArray()))
throw _host.ExceptDecode();
var strm = new MemoryStream(buffer, writable: false);
var loader = new BinaryLoader(_host, new BinaryLoader.Arguments(), strm);
Schema = loader.Schema;
}

// Only load Parquest related data if a file is present. Otherwise, just the Schema is valid.
if (files.Count > 0)
{
_parquetOptions = new ParquetOptions()
{
TreatByteArrayAsString = true,
TreatBigIntegersAsDates = treatBigIntegersAsDates
};

_parquetStream = OpenStream(files);
DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
ReaderOptions readerOptions = new ReaderOptions()
{
Count = 0,
Offset = 0
};
schemaDataSet = ParquetReader.Read(_parquetStream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnsLoaded = InitColumns(schemaDataSet);

@TomFinleyTomFinleyJul 2, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

_columnsLoaded = InitColumns(schemaDataSet); [](start = 16, length = 44)

What happens if you load a schema from the model, but then the parquet loader does not "agree" with that schema? I don't see how this case is handled here. #Closed

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Added schema check.


In reply to: 199612477 [](ancestors = 199612477)

Schema = CreateSchema(_host, _columnsLoaded);
}
else if (Schema == null)
{
throw _host.Except("Parquet loader must be created with one file");
}
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

return host.Apply("Loading Model",
ch => new ParquetLoader(host, ctx, files));
}

/// <summary>
/// Helper function called by the ParquetLoader constructor to initialize the Columns that belong in the Parquet file.
/// Composite data fields are flattened; for example, a Map Field in Parquet is flattened into a Key column and a Value
/// column.
/// </summary>
/// <param name="ch">Communication channel for error reporting.</param>
/// <param name="cols">The array of flattened columns instantiated from the parquet file.</param>
private void InitColumns(IChannel ch, out Column[] cols)
/// <param name="dataSet">The schema data set.</param>
/// <returns>The array of flattened columns instantiated from the parquet file.</returns>
private Column[] InitColumns(DataSet dataSet)
{
cols = null;
List<Column> columnsLoaded = new List<Column>();

foreach (var parquetField in _schemaDataSet.Schema.Fields)
foreach (var parquetField in dataSet.Schema.Fields)
{
FlattenFields(parquetField, ref columnsLoaded, false);
}
cols = columnsLoaded.ToArray();
return columnsLoaded.ToArray();
}

private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable)
Expand DownExpand Up@@ -239,7 +299,7 @@ private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable
}
else
{
throw new InvalidDataException("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
throw _host.ExceptNotSupp("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
}
}

Expand DownExpand Up@@ -326,7 +386,7 @@ private static Stream OpenStream(string filename)

public long? GetRowCount(bool lazy = true)
{
return _schemaDataSet.TotalRowCount;
return _rowCount;
}

public IRowCursor GetRowCursor(Func<int, bool> predicate, IRandom rand = null)
Expand All@@ -353,9 +413,22 @@ public void Save(ModelSaveContext ctx)
// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader

ctx.Writer.Write(_columnChunkReadSize);
ctx.Writer.Write(_parquetOptions.TreatBigIntegersAsDates);

// Save the schema
var noRows = new EmptyDataView(_host, Schema);
var saverArgs = new BinarySaver.Arguments();
saverArgs.Silent = true;
var saver = new BinarySaver(_host, saverArgs);
using (var strm = new MemoryStream())
{
var allColumns = Enumerable.Range(0, Schema.ColumnCount).ToArray();
saver.SaveData(strm, noRows, allColumns);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Is this possible to refactor? It seems inefficient to first save it to a MemoryStream, and then write that memory stream out. Can we just do it in a single step?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In this specific case not necessarily, since the binary saver requires a seekable writable stream. (E.g., it writes data, then seeks back to the header so it can store in the header the offsets of various records in the file.) The repository writer, on the other hand, is based on a zip archive, which AFAIK does not provide seekable writable streams.

ctx.SaveBinaryStream(SchemaCtxName, w => w.WriteByteArray(strm.ToArray()));
}
}

private sealed class Cursor : RootCursorBase, IRowCursor
Expand All@@ -377,6 +450,8 @@ public Cursor(ParquetLoader parent, Func<int, bool> predicate, IRandom rand)
: base(parent._host)
{
Ch.AssertValue(predicate);
Ch.AssertValue(parent._parquetStream);

_loader = parent;
_fileStream = parent._parquetStream;
_parquetConversions = new ParquetConversions(Ch);
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
145 changes: 110 additions & 35 deletions src/Microsoft.ML.Parquet/ParquetLoader.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -12,6 +12,7 @@
using Microsoft.ML.Runtime;
using Microsoft.ML.Runtime.CommandLine;
using Microsoft.ML.Runtime.Data;
using Microsoft.ML.Runtime.Data.IO;
using Microsoft.ML.Runtime.Internal.Utilities;
using Microsoft.ML.Runtime.Model;
using Parquet;
Expand DownExpand Up@@ -88,48 +89,29 @@ public sealed class Arguments
internal const string ShortName = "Parquet";
internal const string ModelSignature = "PARQELDR";

private const string SchemaCtxName = "Schema.idv";

private readonly IHost _host;
private readonly Stream _parquetStream;
private readonly ParquetOptions _parquetOptions;
private readonly int _columnChunkReadSize;
private readonly Column[] _columnsLoaded;
private readonly DataSet _schemaDataSet;
private const int _defaultColumnChunkReadSize = 1000000;

private bool _disposed;
private long? _rowCount;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: ModelSignature,
verWrittenCur: 0x00010001, // Initial
verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does this commented out code provide value? Can it be removed?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

While we generally prefer to avoid checking in commented in code, the version history of past versions is different, since we want to know why we had to bump the version number each time, and we like to have that first version in there. See e.g., the text loader model, which is the most extreme example I am aware of.

verWrittenCur: 0x00010002, // Add Schema to Model Context
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag

Arguments args = new Arguments
{
ColumnChunkReadSize = ctx.Reader.ReadInt32(),
TreatBigIntegersAsDates = ctx.Reader.ReadBoolean()
};
return host.Apply("Loading Model",
ch => new ParquetLoader(args, host, OpenStream(files)));
}

public ParquetLoader(IHostEnvironment env, Arguments args, IMultiStreamSource files)
: this(env, args, OpenStream(files))
{
Expand DownExpand Up@@ -165,6 +147,8 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
TreatBigIntegersAsDates = args.TreatBigIntegersAsDates
};

DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
Expand All@@ -173,36 +157,112 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
Count = 0,
Offset = 0
};
_schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnChunkReadSize = args.ColumnChunkReadSize;
InitColumns(ch, out _columnsLoaded);
_columnsLoaded = InitColumns(schemaDataSet);
Schema = CreateSchema(_host, _columnsLoaded);
}
}

private ParquetLoader(IHost host, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.AssertValue(host);
_host = host;
_host.AssertValue(ctx);
_host.AssertValue(files);

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader (0x00010002)

_columnChunkReadSize = ctx.Reader.ReadInt32();
bool treatBigIntegersAsDates = ctx.Reader.ReadBoolean();

if (ctx.Header.ModelVerWritten >= 0x00010002)
{
// Load the schema
byte[] buffer = null;
if (!ctx.TryLoadBinaryStream(SchemaCtxName, r => buffer = r.ReadByteArray()))
throw _host.ExceptDecode();
var strm = new MemoryStream(buffer, writable: false);
var loader = new BinaryLoader(_host, new BinaryLoader.Arguments(), strm);
Schema = loader.Schema;
}

// Only load Parquest related data if a file is present. Otherwise, just the Schema is valid.
if (files.Count > 0)
{
_parquetOptions = new ParquetOptions()
{
TreatByteArrayAsString = true,
TreatBigIntegersAsDates = treatBigIntegersAsDates
};

_parquetStream = OpenStream(files);
DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
ReaderOptions readerOptions = new ReaderOptions()
{
Count = 0,
Offset = 0
};
schemaDataSet = ParquetReader.Read(_parquetStream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnsLoaded = InitColumns(schemaDataSet);

@TomFinleyTomFinleyJul 2, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

_columnsLoaded = InitColumns(schemaDataSet); [](start = 16, length = 44)

What happens if you load a schema from the model, but then the parquet loader does not "agree" with that schema? I don't see how this case is handled here. #Closed

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Added schema check.


In reply to: 199612477 [](ancestors = 199612477)

Schema = CreateSchema(_host, _columnsLoaded);
}
else if (Schema == null)
{
throw _host.Except("Parquet loader must be created with one file");
}
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

return host.Apply("Loading Model",
ch => new ParquetLoader(host, ctx, files));
}

/// <summary>
/// Helper function called by the ParquetLoader constructor to initialize the Columns that belong in the Parquet file.
/// Composite data fields are flattened; for example, a Map Field in Parquet is flattened into a Key column and a Value
/// column.
/// </summary>
/// <param name="ch">Communication channel for error reporting.</param>
/// <param name="cols">The array of flattened columns instantiated from the parquet file.</param>
private void InitColumns(IChannel ch, out Column[] cols)
/// <param name="dataSet">The schema data set.</param>
/// <returns>The array of flattened columns instantiated from the parquet file.</returns>
private Column[] InitColumns(DataSet dataSet)
{
cols = null;
List<Column> columnsLoaded = new List<Column>();

foreach (var parquetField in _schemaDataSet.Schema.Fields)
foreach (var parquetField in dataSet.Schema.Fields)
{
FlattenFields(parquetField, ref columnsLoaded, false);
}
cols = columnsLoaded.ToArray();
return columnsLoaded.ToArray();
}

private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable)
Expand DownExpand Up@@ -239,7 +299,7 @@ private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable
}
else
{
throw new InvalidDataException("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
throw _host.ExceptNotSupp("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
}
}

Expand DownExpand Up@@ -326,7 +386,7 @@ private static Stream OpenStream(string filename)

public long? GetRowCount(bool lazy = true)
{
return _schemaDataSet.TotalRowCount;
return _rowCount;
}

public IRowCursor GetRowCursor(Func<int, bool> predicate, IRandom rand = null)
Expand All@@ -353,9 +413,22 @@ public void Save(ModelSaveContext ctx)
// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader

ctx.Writer.Write(_columnChunkReadSize);
ctx.Writer.Write(_parquetOptions.TreatBigIntegersAsDates);

// Save the schema
var noRows = new EmptyDataView(_host, Schema);
var saverArgs = new BinarySaver.Arguments();
saverArgs.Silent = true;
var saver = new BinarySaver(_host, saverArgs);
using (var strm = new MemoryStream())
{
var allColumns = Enumerable.Range(0, Schema.ColumnCount).ToArray();
saver.SaveData(strm, noRows, allColumns);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Is this possible to refactor? It seems inefficient to first save it to a MemoryStream, and then write that memory stream out. Can we just do it in a single step?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In this specific case not necessarily, since the binary saver requires a seekable writable stream. (E.g., it writes data, then seeks back to the header so it can store in the header the offsets of various records in the file.) The repository writer, on the other hand, is based on a zip archive, which AFAIK does not provide seekable writable streams.

ctx.SaveBinaryStream(SchemaCtxName, w => w.WriteByteArray(strm.ToArray()));
}
}

private sealed class Cursor : RootCursorBase, IRowCursor
Expand All@@ -377,6 +450,8 @@ public Cursor(ParquetLoader parent, Func<int, bool> predicate, IRandom rand)
: base(parent._host)
{
Ch.AssertValue(predicate);
Ch.AssertValue(parent._parquetStream);

_loader = parent;
_fileStream = parent._parquetStream;
_parquetConversions = new ParquetConversions(Ch);
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
145 changes: 110 additions & 35 deletions src/Microsoft.ML.Parquet/ParquetLoader.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -12,6 +12,7 @@
using Microsoft.ML.Runtime;
using Microsoft.ML.Runtime.CommandLine;
using Microsoft.ML.Runtime.Data;
using Microsoft.ML.Runtime.Data.IO;
using Microsoft.ML.Runtime.Internal.Utilities;
using Microsoft.ML.Runtime.Model;
using Parquet;
Expand DownExpand Up@@ -88,48 +89,29 @@ public sealed class Arguments
internal const string ShortName = "Parquet";
internal const string ModelSignature = "PARQELDR";

private const string SchemaCtxName = "Schema.idv";

private readonly IHost _host;
private readonly Stream _parquetStream;
private readonly ParquetOptions _parquetOptions;
private readonly int _columnChunkReadSize;
private readonly Column[] _columnsLoaded;
private readonly DataSet _schemaDataSet;
private const int _defaultColumnChunkReadSize = 1000000;

private bool _disposed;
private long? _rowCount;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: ModelSignature,
verWrittenCur: 0x00010001, // Initial
verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does this commented out code provide value? Can it be removed?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

While we generally prefer to avoid checking in commented in code, the version history of past versions is different, since we want to know why we had to bump the version number each time, and we like to have that first version in there. See e.g., the text loader model, which is the most extreme example I am aware of.

verWrittenCur: 0x00010002, // Add Schema to Model Context
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag

Arguments args = new Arguments
{
ColumnChunkReadSize = ctx.Reader.ReadInt32(),
TreatBigIntegersAsDates = ctx.Reader.ReadBoolean()
};
return host.Apply("Loading Model",
ch => new ParquetLoader(args, host, OpenStream(files)));
}

public ParquetLoader(IHostEnvironment env, Arguments args, IMultiStreamSource files)
: this(env, args, OpenStream(files))
{
Expand DownExpand Up@@ -165,6 +147,8 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
TreatBigIntegersAsDates = args.TreatBigIntegersAsDates
};

DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
Expand All@@ -173,36 +157,112 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
Count = 0,
Offset = 0
};
_schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnChunkReadSize = args.ColumnChunkReadSize;
InitColumns(ch, out _columnsLoaded);
_columnsLoaded = InitColumns(schemaDataSet);
Schema = CreateSchema(_host, _columnsLoaded);
}
}

private ParquetLoader(IHost host, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.AssertValue(host);
_host = host;
_host.AssertValue(ctx);
_host.AssertValue(files);

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader (0x00010002)

_columnChunkReadSize = ctx.Reader.ReadInt32();
bool treatBigIntegersAsDates = ctx.Reader.ReadBoolean();

if (ctx.Header.ModelVerWritten >= 0x00010002)
{
// Load the schema
byte[] buffer = null;
if (!ctx.TryLoadBinaryStream(SchemaCtxName, r => buffer = r.ReadByteArray()))
throw _host.ExceptDecode();
var strm = new MemoryStream(buffer, writable: false);
var loader = new BinaryLoader(_host, new BinaryLoader.Arguments(), strm);
Schema = loader.Schema;
}

// Only load Parquest related data if a file is present. Otherwise, just the Schema is valid.
if (files.Count > 0)
{
_parquetOptions = new ParquetOptions()
{
TreatByteArrayAsString = true,
TreatBigIntegersAsDates = treatBigIntegersAsDates
};

_parquetStream = OpenStream(files);
DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
ReaderOptions readerOptions = new ReaderOptions()
{
Count = 0,
Offset = 0
};
schemaDataSet = ParquetReader.Read(_parquetStream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnsLoaded = InitColumns(schemaDataSet);

@TomFinleyTomFinleyJul 2, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

_columnsLoaded = InitColumns(schemaDataSet); [](start = 16, length = 44)

What happens if you load a schema from the model, but then the parquet loader does not "agree" with that schema? I don't see how this case is handled here. #Closed

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Added schema check.


In reply to: 199612477 [](ancestors = 199612477)

Schema = CreateSchema(_host, _columnsLoaded);
}
else if (Schema == null)
{
throw _host.Except("Parquet loader must be created with one file");
}
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

return host.Apply("Loading Model",
ch => new ParquetLoader(host, ctx, files));
}

/// <summary>
/// Helper function called by the ParquetLoader constructor to initialize the Columns that belong in the Parquet file.
/// Composite data fields are flattened; for example, a Map Field in Parquet is flattened into a Key column and a Value
/// column.
/// </summary>
/// <param name="ch">Communication channel for error reporting.</param>
/// <param name="cols">The array of flattened columns instantiated from the parquet file.</param>
private void InitColumns(IChannel ch, out Column[] cols)
/// <param name="dataSet">The schema data set.</param>
/// <returns>The array of flattened columns instantiated from the parquet file.</returns>
private Column[] InitColumns(DataSet dataSet)
{
cols = null;
List<Column> columnsLoaded = new List<Column>();

foreach (var parquetField in _schemaDataSet.Schema.Fields)
foreach (var parquetField in dataSet.Schema.Fields)
{
FlattenFields(parquetField, ref columnsLoaded, false);
}
cols = columnsLoaded.ToArray();
return columnsLoaded.ToArray();
}

private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable)
Expand DownExpand Up@@ -239,7 +299,7 @@ private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable
}
else
{
throw new InvalidDataException("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
throw _host.ExceptNotSupp("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
}
}

Expand DownExpand Up@@ -326,7 +386,7 @@ private static Stream OpenStream(string filename)

public long? GetRowCount(bool lazy = true)
{
return _schemaDataSet.TotalRowCount;
return _rowCount;
}

public IRowCursor GetRowCursor(Func<int, bool> predicate, IRandom rand = null)
Expand All@@ -353,9 +413,22 @@ public void Save(ModelSaveContext ctx)
// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader

ctx.Writer.Write(_columnChunkReadSize);
ctx.Writer.Write(_parquetOptions.TreatBigIntegersAsDates);

// Save the schema
var noRows = new EmptyDataView(_host, Schema);
var saverArgs = new BinarySaver.Arguments();
saverArgs.Silent = true;
var saver = new BinarySaver(_host, saverArgs);
using (var strm = new MemoryStream())
{
var allColumns = Enumerable.Range(0, Schema.ColumnCount).ToArray();
saver.SaveData(strm, noRows, allColumns);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Is this possible to refactor? It seems inefficient to first save it to a MemoryStream, and then write that memory stream out. Can we just do it in a single step?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In this specific case not necessarily, since the binary saver requires a seekable writable stream. (E.g., it writes data, then seeks back to the header so it can store in the header the offsets of various records in the file.) The repository writer, on the other hand, is based on a zip archive, which AFAIK does not provide seekable writable streams.

ctx.SaveBinaryStream(SchemaCtxName, w => w.WriteByteArray(strm.ToArray()));
}
}

private sealed class Cursor : RootCursorBase, IRowCursor
Expand All@@ -377,6 +450,8 @@ public Cursor(ParquetLoader parent, Func<int, bool> predicate, IRandom rand)
: base(parent._host)
{
Ch.AssertValue(predicate);
Ch.AssertValue(parent._parquetStream);

_loader = parent;
_fileStream = parent._parquetStream;
_parquetConversions = new ParquetConversions(Ch);
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
145 changes: 110 additions & 35 deletions src/Microsoft.ML.Parquet/ParquetLoader.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -12,6 +12,7 @@
using Microsoft.ML.Runtime;
using Microsoft.ML.Runtime.CommandLine;
using Microsoft.ML.Runtime.Data;
using Microsoft.ML.Runtime.Data.IO;
using Microsoft.ML.Runtime.Internal.Utilities;
using Microsoft.ML.Runtime.Model;
using Parquet;
Expand DownExpand Up@@ -88,48 +89,29 @@ public sealed class Arguments
internal const string ShortName = "Parquet";
internal const string ModelSignature = "PARQELDR";

private const string SchemaCtxName = "Schema.idv";

private readonly IHost _host;
private readonly Stream _parquetStream;
private readonly ParquetOptions _parquetOptions;
private readonly int _columnChunkReadSize;
private readonly Column[] _columnsLoaded;
private readonly DataSet _schemaDataSet;
private const int _defaultColumnChunkReadSize = 1000000;

private bool _disposed;
private long? _rowCount;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: ModelSignature,
verWrittenCur: 0x00010001, // Initial
verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does this commented out code provide value? Can it be removed?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

While we generally prefer to avoid checking in commented in code, the version history of past versions is different, since we want to know why we had to bump the version number each time, and we like to have that first version in there. See e.g., the text loader model, which is the most extreme example I am aware of.

verWrittenCur: 0x00010002, // Add Schema to Model Context
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag

Arguments args = new Arguments
{
ColumnChunkReadSize = ctx.Reader.ReadInt32(),
TreatBigIntegersAsDates = ctx.Reader.ReadBoolean()
};
return host.Apply("Loading Model",
ch => new ParquetLoader(args, host, OpenStream(files)));
}

public ParquetLoader(IHostEnvironment env, Arguments args, IMultiStreamSource files)
: this(env, args, OpenStream(files))
{
Expand DownExpand Up@@ -165,6 +147,8 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
TreatBigIntegersAsDates = args.TreatBigIntegersAsDates
};

DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
Expand All@@ -173,36 +157,112 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
Count = 0,
Offset = 0
};
_schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnChunkReadSize = args.ColumnChunkReadSize;
InitColumns(ch, out _columnsLoaded);
_columnsLoaded = InitColumns(schemaDataSet);
Schema = CreateSchema(_host, _columnsLoaded);
}
}

private ParquetLoader(IHost host, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.AssertValue(host);
_host = host;
_host.AssertValue(ctx);
_host.AssertValue(files);

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader (0x00010002)

_columnChunkReadSize = ctx.Reader.ReadInt32();
bool treatBigIntegersAsDates = ctx.Reader.ReadBoolean();

if (ctx.Header.ModelVerWritten >= 0x00010002)
{
// Load the schema
byte[] buffer = null;
if (!ctx.TryLoadBinaryStream(SchemaCtxName, r => buffer = r.ReadByteArray()))
throw _host.ExceptDecode();
var strm = new MemoryStream(buffer, writable: false);
var loader = new BinaryLoader(_host, new BinaryLoader.Arguments(), strm);
Schema = loader.Schema;
}

// Only load Parquest related data if a file is present. Otherwise, just the Schema is valid.
if (files.Count > 0)
{
_parquetOptions = new ParquetOptions()
{
TreatByteArrayAsString = true,
TreatBigIntegersAsDates = treatBigIntegersAsDates
};

_parquetStream = OpenStream(files);
DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
ReaderOptions readerOptions = new ReaderOptions()
{
Count = 0,
Offset = 0
};
schemaDataSet = ParquetReader.Read(_parquetStream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnsLoaded = InitColumns(schemaDataSet);

@TomFinleyTomFinleyJul 2, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

_columnsLoaded = InitColumns(schemaDataSet); [](start = 16, length = 44)

What happens if you load a schema from the model, but then the parquet loader does not "agree" with that schema? I don't see how this case is handled here. #Closed

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Added schema check.


In reply to: 199612477 [](ancestors = 199612477)

Schema = CreateSchema(_host, _columnsLoaded);
}
else if (Schema == null)
{
throw _host.Except("Parquet loader must be created with one file");
}
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

return host.Apply("Loading Model",
ch => new ParquetLoader(host, ctx, files));
}

/// <summary>
/// Helper function called by the ParquetLoader constructor to initialize the Columns that belong in the Parquet file.
/// Composite data fields are flattened; for example, a Map Field in Parquet is flattened into a Key column and a Value
/// column.
/// </summary>
/// <param name="ch">Communication channel for error reporting.</param>
/// <param name="cols">The array of flattened columns instantiated from the parquet file.</param>
private void InitColumns(IChannel ch, out Column[] cols)
/// <param name="dataSet">The schema data set.</param>
/// <returns>The array of flattened columns instantiated from the parquet file.</returns>
private Column[] InitColumns(DataSet dataSet)
{
cols = null;
List<Column> columnsLoaded = new List<Column>();

foreach (var parquetField in _schemaDataSet.Schema.Fields)
foreach (var parquetField in dataSet.Schema.Fields)
{
FlattenFields(parquetField, ref columnsLoaded, false);
}
cols = columnsLoaded.ToArray();
return columnsLoaded.ToArray();
}

private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable)
Expand DownExpand Up@@ -239,7 +299,7 @@ private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable
}
else
{
throw new InvalidDataException("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
throw _host.ExceptNotSupp("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
}
}

Expand DownExpand Up@@ -326,7 +386,7 @@ private static Stream OpenStream(string filename)

public long? GetRowCount(bool lazy = true)
{
return _schemaDataSet.TotalRowCount;
return _rowCount;
}

public IRowCursor GetRowCursor(Func<int, bool> predicate, IRandom rand = null)
Expand All@@ -353,9 +413,22 @@ public void Save(ModelSaveContext ctx)
// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader

ctx.Writer.Write(_columnChunkReadSize);
ctx.Writer.Write(_parquetOptions.TreatBigIntegersAsDates);

// Save the schema
var noRows = new EmptyDataView(_host, Schema);
var saverArgs = new BinarySaver.Arguments();
saverArgs.Silent = true;
var saver = new BinarySaver(_host, saverArgs);
using (var strm = new MemoryStream())
{
var allColumns = Enumerable.Range(0, Schema.ColumnCount).ToArray();
saver.SaveData(strm, noRows, allColumns);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Is this possible to refactor? It seems inefficient to first save it to a MemoryStream, and then write that memory stream out. Can we just do it in a single step?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In this specific case not necessarily, since the binary saver requires a seekable writable stream. (E.g., it writes data, then seeks back to the header so it can store in the header the offsets of various records in the file.) The repository writer, on the other hand, is based on a zip archive, which AFAIK does not provide seekable writable streams.

ctx.SaveBinaryStream(SchemaCtxName, w => w.WriteByteArray(strm.ToArray()));
}
}

private sealed class Cursor : RootCursorBase, IRowCursor
Expand All@@ -377,6 +450,8 @@ public Cursor(ParquetLoader parent, Func<int, bool> predicate, IRandom rand)
: base(parent._host)
{
Ch.AssertValue(predicate);
Ch.AssertValue(parent._parquetStream);

_loader = parent;
_fileStream = parent._parquetStream;
_parquetConversions = new ParquetConversions(Ch);
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
145 changes: 110 additions & 35 deletions src/Microsoft.ML.Parquet/ParquetLoader.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -12,6 +12,7 @@
using Microsoft.ML.Runtime;
using Microsoft.ML.Runtime.CommandLine;
using Microsoft.ML.Runtime.Data;
using Microsoft.ML.Runtime.Data.IO;
using Microsoft.ML.Runtime.Internal.Utilities;
using Microsoft.ML.Runtime.Model;
using Parquet;
Expand DownExpand Up@@ -88,48 +89,29 @@ public sealed class Arguments
internal const string ShortName = "Parquet";
internal const string ModelSignature = "PARQELDR";

private const string SchemaCtxName = "Schema.idv";

private readonly IHost _host;
private readonly Stream _parquetStream;
private readonly ParquetOptions _parquetOptions;
private readonly int _columnChunkReadSize;
private readonly Column[] _columnsLoaded;
private readonly DataSet _schemaDataSet;
private const int _defaultColumnChunkReadSize = 1000000;

private bool _disposed;
private long? _rowCount;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: ModelSignature,
verWrittenCur: 0x00010001, // Initial
verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does this commented out code provide value? Can it be removed?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

While we generally prefer to avoid checking in commented in code, the version history of past versions is different, since we want to know why we had to bump the version number each time, and we like to have that first version in there. See e.g., the text loader model, which is the most extreme example I am aware of.

verWrittenCur: 0x00010002, // Add Schema to Model Context
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag

Arguments args = new Arguments
{
ColumnChunkReadSize = ctx.Reader.ReadInt32(),
TreatBigIntegersAsDates = ctx.Reader.ReadBoolean()
};
return host.Apply("Loading Model",
ch => new ParquetLoader(args, host, OpenStream(files)));
}

public ParquetLoader(IHostEnvironment env, Arguments args, IMultiStreamSource files)
: this(env, args, OpenStream(files))
{
Expand DownExpand Up@@ -165,6 +147,8 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
TreatBigIntegersAsDates = args.TreatBigIntegersAsDates
};

DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
Expand All@@ -173,36 +157,112 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
Count = 0,
Offset = 0
};
_schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnChunkReadSize = args.ColumnChunkReadSize;
InitColumns(ch, out _columnsLoaded);
_columnsLoaded = InitColumns(schemaDataSet);
Schema = CreateSchema(_host, _columnsLoaded);
}
}

private ParquetLoader(IHost host, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.AssertValue(host);
_host = host;
_host.AssertValue(ctx);
_host.AssertValue(files);

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader (0x00010002)

_columnChunkReadSize = ctx.Reader.ReadInt32();
bool treatBigIntegersAsDates = ctx.Reader.ReadBoolean();

if (ctx.Header.ModelVerWritten >= 0x00010002)
{
// Load the schema
byte[] buffer = null;
if (!ctx.TryLoadBinaryStream(SchemaCtxName, r => buffer = r.ReadByteArray()))
throw _host.ExceptDecode();
var strm = new MemoryStream(buffer, writable: false);
var loader = new BinaryLoader(_host, new BinaryLoader.Arguments(), strm);
Schema = loader.Schema;
}

// Only load Parquest related data if a file is present. Otherwise, just the Schema is valid.
if (files.Count > 0)
{
_parquetOptions = new ParquetOptions()
{
TreatByteArrayAsString = true,
TreatBigIntegersAsDates = treatBigIntegersAsDates
};

_parquetStream = OpenStream(files);
DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
ReaderOptions readerOptions = new ReaderOptions()
{
Count = 0,
Offset = 0
};
schemaDataSet = ParquetReader.Read(_parquetStream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnsLoaded = InitColumns(schemaDataSet);

@TomFinleyTomFinleyJul 2, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

_columnsLoaded = InitColumns(schemaDataSet); [](start = 16, length = 44)

What happens if you load a schema from the model, but then the parquet loader does not "agree" with that schema? I don't see how this case is handled here. #Closed

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Added schema check.


In reply to: 199612477 [](ancestors = 199612477)

Schema = CreateSchema(_host, _columnsLoaded);
}
else if (Schema == null)
{
throw _host.Except("Parquet loader must be created with one file");
}
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

return host.Apply("Loading Model",
ch => new ParquetLoader(host, ctx, files));
}

/// <summary>
/// Helper function called by the ParquetLoader constructor to initialize the Columns that belong in the Parquet file.
/// Composite data fields are flattened; for example, a Map Field in Parquet is flattened into a Key column and a Value
/// column.
/// </summary>
/// <param name="ch">Communication channel for error reporting.</param>
/// <param name="cols">The array of flattened columns instantiated from the parquet file.</param>
private void InitColumns(IChannel ch, out Column[] cols)
/// <param name="dataSet">The schema data set.</param>
/// <returns>The array of flattened columns instantiated from the parquet file.</returns>
private Column[] InitColumns(DataSet dataSet)
{
cols = null;
List<Column> columnsLoaded = new List<Column>();

foreach (var parquetField in _schemaDataSet.Schema.Fields)
foreach (var parquetField in dataSet.Schema.Fields)
{
FlattenFields(parquetField, ref columnsLoaded, false);
}
cols = columnsLoaded.ToArray();
return columnsLoaded.ToArray();
}

private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable)
Expand DownExpand Up@@ -239,7 +299,7 @@ private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable
}
else
{
throw new InvalidDataException("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
throw _host.ExceptNotSupp("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
}
}

Expand DownExpand Up@@ -326,7 +386,7 @@ private static Stream OpenStream(string filename)

public long? GetRowCount(bool lazy = true)
{
return _schemaDataSet.TotalRowCount;
return _rowCount;
}

public IRowCursor GetRowCursor(Func<int, bool> predicate, IRandom rand = null)
Expand All@@ -353,9 +413,22 @@ public void Save(ModelSaveContext ctx)
// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader

ctx.Writer.Write(_columnChunkReadSize);
ctx.Writer.Write(_parquetOptions.TreatBigIntegersAsDates);

// Save the schema
var noRows = new EmptyDataView(_host, Schema);
var saverArgs = new BinarySaver.Arguments();
saverArgs.Silent = true;
var saver = new BinarySaver(_host, saverArgs);
using (var strm = new MemoryStream())
{
var allColumns = Enumerable.Range(0, Schema.ColumnCount).ToArray();
saver.SaveData(strm, noRows, allColumns);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Is this possible to refactor? It seems inefficient to first save it to a MemoryStream, and then write that memory stream out. Can we just do it in a single step?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In this specific case not necessarily, since the binary saver requires a seekable writable stream. (E.g., it writes data, then seeks back to the header so it can store in the header the offsets of various records in the file.) The repository writer, on the other hand, is based on a zip archive, which AFAIK does not provide seekable writable streams.

ctx.SaveBinaryStream(SchemaCtxName, w => w.WriteByteArray(strm.ToArray()));
}
}

private sealed class Cursor : RootCursorBase, IRowCursor
Expand All@@ -377,6 +450,8 @@ public Cursor(ParquetLoader parent, Func<int, bool> predicate, IRandom rand)
: base(parent._host)
{
Ch.AssertValue(predicate);
Ch.AssertValue(parent._parquetStream);

_loader = parent;
_fileStream = parent._parquetStream;
_parquetConversions = new ParquetConversions(Ch);
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
145 changes: 110 additions & 35 deletions src/Microsoft.ML.Parquet/ParquetLoader.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -12,6 +12,7 @@
using Microsoft.ML.Runtime;
using Microsoft.ML.Runtime.CommandLine;
using Microsoft.ML.Runtime.Data;
using Microsoft.ML.Runtime.Data.IO;
using Microsoft.ML.Runtime.Internal.Utilities;
using Microsoft.ML.Runtime.Model;
using Parquet;
Expand DownExpand Up@@ -88,48 +89,29 @@ public sealed class Arguments
internal const string ShortName = "Parquet";
internal const string ModelSignature = "PARQELDR";

private const string SchemaCtxName = "Schema.idv";

private readonly IHost _host;
private readonly Stream _parquetStream;
private readonly ParquetOptions _parquetOptions;
private readonly int _columnChunkReadSize;
private readonly Column[] _columnsLoaded;
private readonly DataSet _schemaDataSet;
private const int _defaultColumnChunkReadSize = 1000000;

private bool _disposed;
private long? _rowCount;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: ModelSignature,
verWrittenCur: 0x00010001, // Initial
verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does this commented out code provide value? Can it be removed?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

While we generally prefer to avoid checking in commented in code, the version history of past versions is different, since we want to know why we had to bump the version number each time, and we like to have that first version in there. See e.g., the text loader model, which is the most extreme example I am aware of.

verWrittenCur: 0x00010002, // Add Schema to Model Context
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag

Arguments args = new Arguments
{
ColumnChunkReadSize = ctx.Reader.ReadInt32(),
TreatBigIntegersAsDates = ctx.Reader.ReadBoolean()
};
return host.Apply("Loading Model",
ch => new ParquetLoader(args, host, OpenStream(files)));
}

public ParquetLoader(IHostEnvironment env, Arguments args, IMultiStreamSource files)
: this(env, args, OpenStream(files))
{
Expand DownExpand Up@@ -165,6 +147,8 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
TreatBigIntegersAsDates = args.TreatBigIntegersAsDates
};

DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
Expand All@@ -173,36 +157,112 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
Count = 0,
Offset = 0
};
_schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnChunkReadSize = args.ColumnChunkReadSize;
InitColumns(ch, out _columnsLoaded);
_columnsLoaded = InitColumns(schemaDataSet);
Schema = CreateSchema(_host, _columnsLoaded);
}
}

private ParquetLoader(IHost host, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.AssertValue(host);
_host = host;
_host.AssertValue(ctx);
_host.AssertValue(files);

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader (0x00010002)

_columnChunkReadSize = ctx.Reader.ReadInt32();
bool treatBigIntegersAsDates = ctx.Reader.ReadBoolean();

if (ctx.Header.ModelVerWritten >= 0x00010002)
{
// Load the schema
byte[] buffer = null;
if (!ctx.TryLoadBinaryStream(SchemaCtxName, r => buffer = r.ReadByteArray()))
throw _host.ExceptDecode();
var strm = new MemoryStream(buffer, writable: false);
var loader = new BinaryLoader(_host, new BinaryLoader.Arguments(), strm);
Schema = loader.Schema;
}

// Only load Parquest related data if a file is present. Otherwise, just the Schema is valid.
if (files.Count > 0)
{
_parquetOptions = new ParquetOptions()
{
TreatByteArrayAsString = true,
TreatBigIntegersAsDates = treatBigIntegersAsDates
};

_parquetStream = OpenStream(files);
DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
ReaderOptions readerOptions = new ReaderOptions()
{
Count = 0,
Offset = 0
};
schemaDataSet = ParquetReader.Read(_parquetStream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnsLoaded = InitColumns(schemaDataSet);

@TomFinleyTomFinleyJul 2, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

_columnsLoaded = InitColumns(schemaDataSet); [](start = 16, length = 44)

What happens if you load a schema from the model, but then the parquet loader does not "agree" with that schema? I don't see how this case is handled here. #Closed

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Added schema check.


In reply to: 199612477 [](ancestors = 199612477)

Schema = CreateSchema(_host, _columnsLoaded);
}
else if (Schema == null)
{
throw _host.Except("Parquet loader must be created with one file");
}
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

return host.Apply("Loading Model",
ch => new ParquetLoader(host, ctx, files));
}

/// <summary>
/// Helper function called by the ParquetLoader constructor to initialize the Columns that belong in the Parquet file.
/// Composite data fields are flattened; for example, a Map Field in Parquet is flattened into a Key column and a Value
/// column.
/// </summary>
/// <param name="ch">Communication channel for error reporting.</param>
/// <param name="cols">The array of flattened columns instantiated from the parquet file.</param>
private void InitColumns(IChannel ch, out Column[] cols)
/// <param name="dataSet">The schema data set.</param>
/// <returns>The array of flattened columns instantiated from the parquet file.</returns>
private Column[] InitColumns(DataSet dataSet)
{
cols = null;
List<Column> columnsLoaded = new List<Column>();

foreach (var parquetField in _schemaDataSet.Schema.Fields)
foreach (var parquetField in dataSet.Schema.Fields)
{
FlattenFields(parquetField, ref columnsLoaded, false);
}
cols = columnsLoaded.ToArray();
return columnsLoaded.ToArray();
}

private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable)
Expand DownExpand Up@@ -239,7 +299,7 @@ private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable
}
else
{
throw new InvalidDataException("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
throw _host.ExceptNotSupp("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
}
}

Expand DownExpand Up@@ -326,7 +386,7 @@ private static Stream OpenStream(string filename)

public long? GetRowCount(bool lazy = true)
{
return _schemaDataSet.TotalRowCount;
return _rowCount;
}

public IRowCursor GetRowCursor(Func<int, bool> predicate, IRandom rand = null)
Expand All@@ -353,9 +413,22 @@ public void Save(ModelSaveContext ctx)
// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader

ctx.Writer.Write(_columnChunkReadSize);
ctx.Writer.Write(_parquetOptions.TreatBigIntegersAsDates);

// Save the schema
var noRows = new EmptyDataView(_host, Schema);
var saverArgs = new BinarySaver.Arguments();
saverArgs.Silent = true;
var saver = new BinarySaver(_host, saverArgs);
using (var strm = new MemoryStream())
{
var allColumns = Enumerable.Range(0, Schema.ColumnCount).ToArray();
saver.SaveData(strm, noRows, allColumns);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Is this possible to refactor? It seems inefficient to first save it to a MemoryStream, and then write that memory stream out. Can we just do it in a single step?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In this specific case not necessarily, since the binary saver requires a seekable writable stream. (E.g., it writes data, then seeks back to the header so it can store in the header the offsets of various records in the file.) The repository writer, on the other hand, is based on a zip archive, which AFAIK does not provide seekable writable streams.

ctx.SaveBinaryStream(SchemaCtxName, w => w.WriteByteArray(strm.ToArray()));
}
}

private sealed class Cursor : RootCursorBase, IRowCursor
Expand All@@ -377,6 +450,8 @@ public Cursor(ParquetLoader parent, Func<int, bool> predicate, IRandom rand)
: base(parent._host)
{
Ch.AssertValue(predicate);
Ch.AssertValue(parent._parquetStream);

_loader = parent;
_fileStream = parent._parquetStream;
_parquetConversions = new ParquetConversions(Ch);
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
145 changes: 110 additions & 35 deletions src/Microsoft.ML.Parquet/ParquetLoader.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -12,6 +12,7 @@
using Microsoft.ML.Runtime;
using Microsoft.ML.Runtime.CommandLine;
using Microsoft.ML.Runtime.Data;
using Microsoft.ML.Runtime.Data.IO;
using Microsoft.ML.Runtime.Internal.Utilities;
using Microsoft.ML.Runtime.Model;
using Parquet;
Expand DownExpand Up@@ -88,48 +89,29 @@ public sealed class Arguments
internal const string ShortName = "Parquet";
internal const string ModelSignature = "PARQELDR";

private const string SchemaCtxName = "Schema.idv";

private readonly IHost _host;
private readonly Stream _parquetStream;
private readonly ParquetOptions _parquetOptions;
private readonly int _columnChunkReadSize;
private readonly Column[] _columnsLoaded;
private readonly DataSet _schemaDataSet;
private const int _defaultColumnChunkReadSize = 1000000;

private bool _disposed;
private long? _rowCount;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: ModelSignature,
verWrittenCur: 0x00010001, // Initial
verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does this commented out code provide value? Can it be removed?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

While we generally prefer to avoid checking in commented in code, the version history of past versions is different, since we want to know why we had to bump the version number each time, and we like to have that first version in there. See e.g., the text loader model, which is the most extreme example I am aware of.

verWrittenCur: 0x00010002, // Add Schema to Model Context
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag

Arguments args = new Arguments
{
ColumnChunkReadSize = ctx.Reader.ReadInt32(),
TreatBigIntegersAsDates = ctx.Reader.ReadBoolean()
};
return host.Apply("Loading Model",
ch => new ParquetLoader(args, host, OpenStream(files)));
}

public ParquetLoader(IHostEnvironment env, Arguments args, IMultiStreamSource files)
: this(env, args, OpenStream(files))
{
Expand DownExpand Up@@ -165,6 +147,8 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
TreatBigIntegersAsDates = args.TreatBigIntegersAsDates
};

DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
Expand All@@ -173,36 +157,112 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
Count = 0,
Offset = 0
};
_schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnChunkReadSize = args.ColumnChunkReadSize;
InitColumns(ch, out _columnsLoaded);
_columnsLoaded = InitColumns(schemaDataSet);
Schema = CreateSchema(_host, _columnsLoaded);
}
}

private ParquetLoader(IHost host, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.AssertValue(host);
_host = host;
_host.AssertValue(ctx);
_host.AssertValue(files);

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader (0x00010002)

_columnChunkReadSize = ctx.Reader.ReadInt32();
bool treatBigIntegersAsDates = ctx.Reader.ReadBoolean();

if (ctx.Header.ModelVerWritten >= 0x00010002)
{
// Load the schema
byte[] buffer = null;
if (!ctx.TryLoadBinaryStream(SchemaCtxName, r => buffer = r.ReadByteArray()))
throw _host.ExceptDecode();
var strm = new MemoryStream(buffer, writable: false);
var loader = new BinaryLoader(_host, new BinaryLoader.Arguments(), strm);
Schema = loader.Schema;
}

// Only load Parquest related data if a file is present. Otherwise, just the Schema is valid.
if (files.Count > 0)
{
_parquetOptions = new ParquetOptions()
{
TreatByteArrayAsString = true,
TreatBigIntegersAsDates = treatBigIntegersAsDates
};

_parquetStream = OpenStream(files);
DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
ReaderOptions readerOptions = new ReaderOptions()
{
Count = 0,
Offset = 0
};
schemaDataSet = ParquetReader.Read(_parquetStream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnsLoaded = InitColumns(schemaDataSet);

@TomFinleyTomFinleyJul 2, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

_columnsLoaded = InitColumns(schemaDataSet); [](start = 16, length = 44)

What happens if you load a schema from the model, but then the parquet loader does not "agree" with that schema? I don't see how this case is handled here. #Closed

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Added schema check.


In reply to: 199612477 [](ancestors = 199612477)

Schema = CreateSchema(_host, _columnsLoaded);
}
else if (Schema == null)
{
throw _host.Except("Parquet loader must be created with one file");
}
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

return host.Apply("Loading Model",
ch => new ParquetLoader(host, ctx, files));
}

/// <summary>
/// Helper function called by the ParquetLoader constructor to initialize the Columns that belong in the Parquet file.
/// Composite data fields are flattened; for example, a Map Field in Parquet is flattened into a Key column and a Value
/// column.
/// </summary>
/// <param name="ch">Communication channel for error reporting.</param>
/// <param name="cols">The array of flattened columns instantiated from the parquet file.</param>
private void InitColumns(IChannel ch, out Column[] cols)
/// <param name="dataSet">The schema data set.</param>
/// <returns>The array of flattened columns instantiated from the parquet file.</returns>
private Column[] InitColumns(DataSet dataSet)
{
cols = null;
List<Column> columnsLoaded = new List<Column>();

foreach (var parquetField in _schemaDataSet.Schema.Fields)
foreach (var parquetField in dataSet.Schema.Fields)
{
FlattenFields(parquetField, ref columnsLoaded, false);
}
cols = columnsLoaded.ToArray();
return columnsLoaded.ToArray();
}

private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable)
Expand DownExpand Up@@ -239,7 +299,7 @@ private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable
}
else
{
throw new InvalidDataException("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
throw _host.ExceptNotSupp("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
}
}

Expand DownExpand Up@@ -326,7 +386,7 @@ private static Stream OpenStream(string filename)

public long? GetRowCount(bool lazy = true)
{
return _schemaDataSet.TotalRowCount;
return _rowCount;
}

public IRowCursor GetRowCursor(Func<int, bool> predicate, IRandom rand = null)
Expand All@@ -353,9 +413,22 @@ public void Save(ModelSaveContext ctx)
// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader

ctx.Writer.Write(_columnChunkReadSize);
ctx.Writer.Write(_parquetOptions.TreatBigIntegersAsDates);

// Save the schema
var noRows = new EmptyDataView(_host, Schema);
var saverArgs = new BinarySaver.Arguments();
saverArgs.Silent = true;
var saver = new BinarySaver(_host, saverArgs);
using (var strm = new MemoryStream())
{
var allColumns = Enumerable.Range(0, Schema.ColumnCount).ToArray();
saver.SaveData(strm, noRows, allColumns);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Is this possible to refactor? It seems inefficient to first save it to a MemoryStream, and then write that memory stream out. Can we just do it in a single step?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In this specific case not necessarily, since the binary saver requires a seekable writable stream. (E.g., it writes data, then seeks back to the header so it can store in the header the offsets of various records in the file.) The repository writer, on the other hand, is based on a zip archive, which AFAIK does not provide seekable writable streams.

ctx.SaveBinaryStream(SchemaCtxName, w => w.WriteByteArray(strm.ToArray()));
}
}

private sealed class Cursor : RootCursorBase, IRowCursor
Expand All@@ -377,6 +450,8 @@ public Cursor(ParquetLoader parent, Func<int, bool> predicate, IRandom rand)
: base(parent._host)
{
Ch.AssertValue(predicate);
Ch.AssertValue(parent._parquetStream);

_loader = parent;
_fileStream = parent._parquetStream;
_parquetConversions = new ParquetConversions(Ch);
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
145 changes: 110 additions & 35 deletions src/Microsoft.ML.Parquet/ParquetLoader.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -12,6 +12,7 @@
using Microsoft.ML.Runtime;
using Microsoft.ML.Runtime.CommandLine;
using Microsoft.ML.Runtime.Data;
using Microsoft.ML.Runtime.Data.IO;
using Microsoft.ML.Runtime.Internal.Utilities;
using Microsoft.ML.Runtime.Model;
using Parquet;
Expand DownExpand Up@@ -88,48 +89,29 @@ public sealed class Arguments
internal const string ShortName = "Parquet";
internal const string ModelSignature = "PARQELDR";

private const string SchemaCtxName = "Schema.idv";

private readonly IHost _host;
private readonly Stream _parquetStream;
private readonly ParquetOptions _parquetOptions;
private readonly int _columnChunkReadSize;
private readonly Column[] _columnsLoaded;
private readonly DataSet _schemaDataSet;
private const int _defaultColumnChunkReadSize = 1000000;

private bool _disposed;
private long? _rowCount;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: ModelSignature,
verWrittenCur: 0x00010001, // Initial
verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does this commented out code provide value? Can it be removed?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

While we generally prefer to avoid checking in commented in code, the version history of past versions is different, since we want to know why we had to bump the version number each time, and we like to have that first version in there. See e.g., the text loader model, which is the most extreme example I am aware of.

verWrittenCur: 0x00010002, // Add Schema to Model Context
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag

Arguments args = new Arguments
{
ColumnChunkReadSize = ctx.Reader.ReadInt32(),
TreatBigIntegersAsDates = ctx.Reader.ReadBoolean()
};
return host.Apply("Loading Model",
ch => new ParquetLoader(args, host, OpenStream(files)));
}

public ParquetLoader(IHostEnvironment env, Arguments args, IMultiStreamSource files)
: this(env, args, OpenStream(files))
{
Expand DownExpand Up@@ -165,6 +147,8 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
TreatBigIntegersAsDates = args.TreatBigIntegersAsDates
};

DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
Expand All@@ -173,36 +157,112 @@ private ParquetLoader(Arguments args, IHost host, Stream stream)
Count = 0,
Offset = 0
};
_schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
schemaDataSet = ParquetReader.Read(stream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnChunkReadSize = args.ColumnChunkReadSize;
InitColumns(ch, out _columnsLoaded);
_columnsLoaded = InitColumns(schemaDataSet);
Schema = CreateSchema(_host, _columnsLoaded);
}
}

private ParquetLoader(IHost host, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.AssertValue(host);
_host = host;
_host.AssertValue(ctx);
_host.AssertValue(files);

// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader (0x00010002)

_columnChunkReadSize = ctx.Reader.ReadInt32();
bool treatBigIntegersAsDates = ctx.Reader.ReadBoolean();

if (ctx.Header.ModelVerWritten >= 0x00010002)
{
// Load the schema
byte[] buffer = null;
if (!ctx.TryLoadBinaryStream(SchemaCtxName, r => buffer = r.ReadByteArray()))
throw _host.ExceptDecode();
var strm = new MemoryStream(buffer, writable: false);
var loader = new BinaryLoader(_host, new BinaryLoader.Arguments(), strm);
Schema = loader.Schema;
}

// Only load Parquest related data if a file is present. Otherwise, just the Schema is valid.
if (files.Count > 0)
{
_parquetOptions = new ParquetOptions()
{
TreatByteArrayAsString = true,
TreatBigIntegersAsDates = treatBigIntegersAsDates
};

_parquetStream = OpenStream(files);
DataSet schemaDataSet;

try
{
// We only care about the schema so ignore the rows.
ReaderOptions readerOptions = new ReaderOptions()
{
Count = 0,
Offset = 0
};
schemaDataSet = ParquetReader.Read(_parquetStream, _parquetOptions, readerOptions);
_rowCount = schemaDataSet.TotalRowCount;
}
catch (Exception ex)
{
throw new InvalidDataException("Cannot read Parquet file", ex);
}

_columnsLoaded = InitColumns(schemaDataSet);

@TomFinleyTomFinleyJul 2, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

_columnsLoaded = InitColumns(schemaDataSet); [](start = 16, length = 44)

What happens if you load a schema from the model, but then the parquet loader does not "agree" with that schema? I don't see how this case is handled here. #Closed

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Added schema check.


In reply to: 199612477 [](ancestors = 199612477)

Schema = CreateSchema(_host, _columnsLoaded);
}
else if (Schema == null)
{
throw _host.Except("Parquet loader must be created with one file");
}
}

public static ParquetLoader Create(IHostEnvironment env, ModelLoadContext ctx, IMultiStreamSource files)
{
Contracts.CheckValue(env, nameof(env));
IHost host = env.Register(LoaderName);

env.CheckValue(ctx, nameof(ctx));
ctx.CheckAtModel(GetVersionInfo());
env.CheckValue(files, nameof(files));

return host.Apply("Loading Model",
ch => new ParquetLoader(host, ctx, files));
}

/// <summary>
/// Helper function called by the ParquetLoader constructor to initialize the Columns that belong in the Parquet file.
/// Composite data fields are flattened; for example, a Map Field in Parquet is flattened into a Key column and a Value
/// column.
/// </summary>
/// <param name="ch">Communication channel for error reporting.</param>
/// <param name="cols">The array of flattened columns instantiated from the parquet file.</param>
private void InitColumns(IChannel ch, out Column[] cols)
/// <param name="dataSet">The schema data set.</param>
/// <returns>The array of flattened columns instantiated from the parquet file.</returns>
private Column[] InitColumns(DataSet dataSet)
{
cols = null;
List<Column> columnsLoaded = new List<Column>();

foreach (var parquetField in _schemaDataSet.Schema.Fields)
foreach (var parquetField in dataSet.Schema.Fields)
{
FlattenFields(parquetField, ref columnsLoaded, false);
}
cols = columnsLoaded.ToArray();
return columnsLoaded.ToArray();
}

private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable)
Expand DownExpand Up@@ -239,7 +299,7 @@ private void FlattenFields(Field field, ref List<Column> cols, bool isRepeatable
}
else
{
throw new InvalidDataException("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
throw _host.ExceptNotSupp("Encountered unknown Parquet field type(Currently recognizes data, map, list, and struct).");
}
}

Expand DownExpand Up@@ -326,7 +386,7 @@ private static Stream OpenStream(string filename)

public long? GetRowCount(bool lazy = true)
{
return _schemaDataSet.TotalRowCount;
return _rowCount;
}

public IRowCursor GetRowCursor(Func<int, bool> predicate, IRandom rand = null)
Expand All@@ -353,9 +413,22 @@ public void Save(ModelSaveContext ctx)
// *** Binary format ***
// int: cached chunk size
// bool: TreatBigIntegersAsDates flag
// Schema of the loader

ctx.Writer.Write(_columnChunkReadSize);
ctx.Writer.Write(_parquetOptions.TreatBigIntegersAsDates);

// Save the schema
var noRows = new EmptyDataView(_host, Schema);
var saverArgs = new BinarySaver.Arguments();
saverArgs.Silent = true;
var saver = new BinarySaver(_host, saverArgs);
using (var strm = new MemoryStream())
{
var allColumns = Enumerable.Range(0, Schema.ColumnCount).ToArray();
saver.SaveData(strm, noRows, allColumns);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Is this possible to refactor? It seems inefficient to first save it to a MemoryStream, and then write that memory stream out. Can we just do it in a single step?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In this specific case not necessarily, since the binary saver requires a seekable writable stream. (E.g., it writes data, then seeks back to the header so it can store in the header the offsets of various records in the file.) The repository writer, on the other hand, is based on a zip archive, which AFAIK does not provide seekable writable streams.

ctx.SaveBinaryStream(SchemaCtxName, w => w.WriteByteArray(strm.ToArray()));
}
}

private sealed class Cursor : RootCursorBase, IRowCursor
Expand All@@ -377,6 +450,8 @@ public Cursor(ParquetLoader parent, Func<int, bool> predicate, IRandom rand)
: base(parent._host)
{
Ch.AssertValue(predicate);
Ch.AssertValue(parent._parquetStream);

_loader = parent;
_fileStream = parent._parquetStream;
_parquetConversions = new ParquetConversions(Ch);
Expand Down