Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
77 changes: 73 additions & 4 deletions src/Microsoft.ML.Transforms/Text/CharTokenizeTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -64,12 +64,16 @@ public sealed class Arguments : TransformInputBase
public const string LoaderSignature = "CharToken";
public const string UserName = "Character Tokenizer Transform";

// Keep track of the model that was saved with ver:0x00010001
private readonly bool _isSeparatorStartEnd;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: "CHARTOKN",
verWrittenCur: 0x00010001, // Initial

@TomFinleyTomFinleyJul 21, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

verWrittenCur: 0x00010001, // Initial [](start = 16, length = 37)

See other examples. You will see that we keep commented out the version strings from old versions, to track them and so we know why we bumped the version each time. This is important: we have live code in our deserializers that is meant to handle those version bumps, and for that reason we like to have documentation on what exactly changed, so that when we have tests on this or that version in our deserializers, we know why they changed. #Resolved

verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial
verWrittenCur: 0x00010002, // Updated to use UnitSeparator <US> character instead of using <ETX><STX> for vector inputs.
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}
Expand All@@ -84,6 +88,7 @@ private static VersionInfo GetVersionInfo()
private volatile string _keyValuesStr;
private volatile int[] _keyValuesBoundaries;

private const ushort UnitSeparator = 0x1f;
private const ushort TextStartMarker = 0x02;
private const ushort TextEndMarker = 0x03;
private const int TextMarkersCount = 2;
Expand DownExpand Up@@ -120,6 +125,8 @@ private CharTokenizeTransform(IHost host, ModelLoadContext ctx, IDataView input)
// byte: _useMarkerChars value.
_useMarkerChars = ctx.Reader.ReadBoolByte();

_isSeparatorStartEnd = ctx.Header.ModelVerReadable < 0x00010002 || ctx.Reader.ReadBoolByte();

_type = GetOutputColumnType();
SetMetadata();
}
Expand All@@ -145,6 +152,7 @@ public override void Save(ModelSaveContext ctx)
// byte: _useMarkerChars value.
SaveBase(ctx);
ctx.Writer.WriteBoolByte(_useMarkerChars);
ctx.Writer.WriteBoolByte(_isSeparatorStartEnd);
}

protected override ColumnType GetColumnTypeCore(int iinfo)
Expand DownExpand Up@@ -399,8 +407,8 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

var getSrc = GetSrcGetter<VBuffer<DvText>>(input, iinfo);
var src = default(VBuffer<DvText>);
return
(ref VBuffer<ushort> dst) =>

ValueGetter<VBuffer<ushort>> getterWithStartEndSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

Expand DownExpand Up@@ -438,6 +446,67 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

dst = new VBuffer<ushort>(len, values, dst.Indices);
};

ValueGetter < VBuffer<ushort> > getterWithUnitSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

int len = 0;

for (int i = 0; i < src.Count; i++)
{
if (src.Values[i].HasChars)
{
len += src.Values[i].Length;

if (i > 0)
len += 1; // add UnitSeparator character to len that will be added
}
}

if (_useMarkerChars)
len += TextMarkersCount;

var values = dst.Values;
if (len > 0)
{
if (Utils.Size(values) < len)
values = new ushort[len];

int index = 0;

// VBuffer<DvText> can be a result of either concatenating text columns together
// or application of word tokenizer before char tokenizer in TextTransform.
//
// Considering VBuffer<DvText> as a single text stream.
// Therefore, prepend and append start and end markers only once i.e. at the start and at end of vector.
// Insert UnitSeparator after every piece of text in the vector.
if (_useMarkerChars)
values[index++] = TextStartMarker;

for (int i = 0; i < src.Count; i++)
{
if (!src.Values[i].HasChars)
continue;

if (i > 0)
values[index++] = UnitSeparator;

for (int ich = 0; ich < src.Values[i].Length; ich++)
{
values[index++] = src.Values[i][ich];
}
}

if (_useMarkerChars)
values[index++] = TextEndMarker;

Contracts.Assert(index == len);
}

dst = new VBuffer<ushort>(len, values, dst.Indices);
};
return _isSeparatorStartEnd ? getterWithStartEndSep : getterWithUnitSep;
}
}
}
54 changes: 25 additions & 29 deletions src/Microsoft.ML.Transforms/Text/TextTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -262,6 +262,30 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new ConcatTransform(h, new ConcatTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)

@zeahmedzeahmedJul 19, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@justinormont, I moved TextNormalizer above WordTokenizer. Let me know if there is any adverse effect of it? #Resolved

@justinormontjustinormontJul 20, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't know if there are averse effects of running TextNormalizer before WordTokenizer. I expect it is different, but assume not adversely so. For instance (depending on the specifics of the tokenizer), I don't can be split into {I, do, n't} but if removing punctuation first, it may be split to {I, dont}.

If you make me a build, I'll run a manual regression test. Unfortunately we don't currently have running nightly regression tests to tell us if the text datasets decreased in their core metrics.

#Resolved

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't like this change. It's far from clear to me that if someone asks to remove stopwords that they meant for the content of stopwords to be retained in the chargrams. At the very least this ought to be a configurable option.


In reply to: 204000786 [](ancestors = 204000786)

@zeahmedzeahmedJul 21, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@TomFinley, Can you elaborate more on this? Your point is not clear to me. Don't you see it good to have TextNormalizer applied before WordTokenizer?


In reply to: 204201054 [](ancestors = 204201054,204000786)

@TomFinleyTomFinleyJul 25, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sorry thought you were saying something else for some reason got confused. #Resolved

{
var xfCols = new TextNormalizerCol[textCols.Length];
string[] dstCols = new string[textCols.Length];
for (int i = 0; i < textCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, textCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = textCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

textCols = dstCols;
}

if (tparams.NeedsWordTokenizationTransform)
{
var xfCols = new DelimitedTokenizeTransform.Column[textCols.Length];
Expand All@@ -281,34 +305,6 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new DelimitedTokenizeTransform(h, new DelimitedTokenizeTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)
{
string[] srcCols = wordTokCols == null ? textCols : wordTokCols;
var xfCols = new TextNormalizerCol[srcCols.Length];
string[] dstCols = new string[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, srcCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = srcCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

if (wordTokCols != null)
wordTokCols = dstCols;
else
textCols = dstCols;
}

if (tparams.NeedsRemoveStopwordsTransform)
{
Contracts.Assert(wordTokCols != null, "StopWords transform requires that word tokenization has been applied to the input text.");
Expand DownExpand Up@@ -360,7 +356,7 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
if (tparams.CharExtractorFactory != null)
{
{
var srcCols = wordTokCols ?? textCols;
var srcCols = tparams.NeedsRemoveStopwordsTransform ? wordTokCols : textCols;
charTokCols = new string[srcCols.Length];
var xfCols = new CharTokenizeTransform.Column[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
Expand Down
2 changes: 1 addition & 1 deletion test/Microsoft.ML.Predictor.Tests/TestPipelineSweeper.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -308,7 +308,7 @@ public void PipelineSweeperRoles()
var trainAuc = bestPipeline.PerformanceSummary.TrainingMetricValue;
var testAuc = bestPipeline.PerformanceSummary.MetricValue;
Assert.True((0.94 < trainAuc) && (trainAuc < 0.95));
Assert.True((0.83 < testAuc) && (testAuc < 0.84));
Assert.True((0.815 < testAuc) && (testAuc < 0.825));

var results = runner.GetOutput<IDataView>("ResultsOut");
Assert.NotNull(results);
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
77 changes: 73 additions & 4 deletions src/Microsoft.ML.Transforms/Text/CharTokenizeTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -64,12 +64,16 @@ public sealed class Arguments : TransformInputBase
public const string LoaderSignature = "CharToken";
public const string UserName = "Character Tokenizer Transform";

// Keep track of the model that was saved with ver:0x00010001
private readonly bool _isSeparatorStartEnd;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: "CHARTOKN",
verWrittenCur: 0x00010001, // Initial

@TomFinleyTomFinleyJul 21, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

verWrittenCur: 0x00010001, // Initial [](start = 16, length = 37)

See other examples. You will see that we keep commented out the version strings from old versions, to track them and so we know why we bumped the version each time. This is important: we have live code in our deserializers that is meant to handle those version bumps, and for that reason we like to have documentation on what exactly changed, so that when we have tests on this or that version in our deserializers, we know why they changed. #Resolved

verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial
verWrittenCur: 0x00010002, // Updated to use UnitSeparator <US> character instead of using <ETX><STX> for vector inputs.
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}
Expand All@@ -84,6 +88,7 @@ private static VersionInfo GetVersionInfo()
private volatile string _keyValuesStr;
private volatile int[] _keyValuesBoundaries;

private const ushort UnitSeparator = 0x1f;
private const ushort TextStartMarker = 0x02;
private const ushort TextEndMarker = 0x03;
private const int TextMarkersCount = 2;
Expand DownExpand Up@@ -120,6 +125,8 @@ private CharTokenizeTransform(IHost host, ModelLoadContext ctx, IDataView input)
// byte: _useMarkerChars value.
_useMarkerChars = ctx.Reader.ReadBoolByte();

_isSeparatorStartEnd = ctx.Header.ModelVerReadable < 0x00010002 || ctx.Reader.ReadBoolByte();

_type = GetOutputColumnType();
SetMetadata();
}
Expand All@@ -145,6 +152,7 @@ public override void Save(ModelSaveContext ctx)
// byte: _useMarkerChars value.
SaveBase(ctx);
ctx.Writer.WriteBoolByte(_useMarkerChars);
ctx.Writer.WriteBoolByte(_isSeparatorStartEnd);
}

protected override ColumnType GetColumnTypeCore(int iinfo)
Expand DownExpand Up@@ -399,8 +407,8 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

var getSrc = GetSrcGetter<VBuffer<DvText>>(input, iinfo);
var src = default(VBuffer<DvText>);
return
(ref VBuffer<ushort> dst) =>

ValueGetter<VBuffer<ushort>> getterWithStartEndSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

Expand DownExpand Up@@ -438,6 +446,67 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

dst = new VBuffer<ushort>(len, values, dst.Indices);
};

ValueGetter < VBuffer<ushort> > getterWithUnitSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

int len = 0;

for (int i = 0; i < src.Count; i++)
{
if (src.Values[i].HasChars)
{
len += src.Values[i].Length;

if (i > 0)
len += 1; // add UnitSeparator character to len that will be added
}
}

if (_useMarkerChars)
len += TextMarkersCount;

var values = dst.Values;
if (len > 0)
{
if (Utils.Size(values) < len)
values = new ushort[len];

int index = 0;

// VBuffer<DvText> can be a result of either concatenating text columns together
// or application of word tokenizer before char tokenizer in TextTransform.
//
// Considering VBuffer<DvText> as a single text stream.
// Therefore, prepend and append start and end markers only once i.e. at the start and at end of vector.
// Insert UnitSeparator after every piece of text in the vector.
if (_useMarkerChars)
values[index++] = TextStartMarker;

for (int i = 0; i < src.Count; i++)
{
if (!src.Values[i].HasChars)
continue;

if (i > 0)
values[index++] = UnitSeparator;

for (int ich = 0; ich < src.Values[i].Length; ich++)
{
values[index++] = src.Values[i][ich];
}
}

if (_useMarkerChars)
values[index++] = TextEndMarker;

Contracts.Assert(index == len);
}

dst = new VBuffer<ushort>(len, values, dst.Indices);
};
return _isSeparatorStartEnd ? getterWithStartEndSep : getterWithUnitSep;
}
}
}
54 changes: 25 additions & 29 deletions src/Microsoft.ML.Transforms/Text/TextTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -262,6 +262,30 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new ConcatTransform(h, new ConcatTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)

@zeahmedzeahmedJul 19, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@justinormont, I moved TextNormalizer above WordTokenizer. Let me know if there is any adverse effect of it? #Resolved

@justinormontjustinormontJul 20, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't know if there are averse effects of running TextNormalizer before WordTokenizer. I expect it is different, but assume not adversely so. For instance (depending on the specifics of the tokenizer), I don't can be split into {I, do, n't} but if removing punctuation first, it may be split to {I, dont}.

If you make me a build, I'll run a manual regression test. Unfortunately we don't currently have running nightly regression tests to tell us if the text datasets decreased in their core metrics.

#Resolved

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't like this change. It's far from clear to me that if someone asks to remove stopwords that they meant for the content of stopwords to be retained in the chargrams. At the very least this ought to be a configurable option.


In reply to: 204000786 [](ancestors = 204000786)

@zeahmedzeahmedJul 21, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@TomFinley, Can you elaborate more on this? Your point is not clear to me. Don't you see it good to have TextNormalizer applied before WordTokenizer?


In reply to: 204201054 [](ancestors = 204201054,204000786)

@TomFinleyTomFinleyJul 25, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sorry thought you were saying something else for some reason got confused. #Resolved

{
var xfCols = new TextNormalizerCol[textCols.Length];
string[] dstCols = new string[textCols.Length];
for (int i = 0; i < textCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, textCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = textCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

textCols = dstCols;
}

if (tparams.NeedsWordTokenizationTransform)
{
var xfCols = new DelimitedTokenizeTransform.Column[textCols.Length];
Expand All@@ -281,34 +305,6 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new DelimitedTokenizeTransform(h, new DelimitedTokenizeTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)
{
string[] srcCols = wordTokCols == null ? textCols : wordTokCols;
var xfCols = new TextNormalizerCol[srcCols.Length];
string[] dstCols = new string[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, srcCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = srcCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

if (wordTokCols != null)
wordTokCols = dstCols;
else
textCols = dstCols;
}

if (tparams.NeedsRemoveStopwordsTransform)
{
Contracts.Assert(wordTokCols != null, "StopWords transform requires that word tokenization has been applied to the input text.");
Expand DownExpand Up@@ -360,7 +356,7 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
if (tparams.CharExtractorFactory != null)
{
{
var srcCols = wordTokCols ?? textCols;
var srcCols = tparams.NeedsRemoveStopwordsTransform ? wordTokCols : textCols;
charTokCols = new string[srcCols.Length];
var xfCols = new CharTokenizeTransform.Column[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
Expand Down
2 changes: 1 addition & 1 deletion test/Microsoft.ML.Predictor.Tests/TestPipelineSweeper.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -308,7 +308,7 @@ public void PipelineSweeperRoles()
var trainAuc = bestPipeline.PerformanceSummary.TrainingMetricValue;
var testAuc = bestPipeline.PerformanceSummary.MetricValue;
Assert.True((0.94 < trainAuc) && (trainAuc < 0.95));
Assert.True((0.83 < testAuc) && (testAuc < 0.84));
Assert.True((0.815 < testAuc) && (testAuc < 0.825));

var results = runner.GetOutput<IDataView>("ResultsOut");
Assert.NotNull(results);
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
77 changes: 73 additions & 4 deletions src/Microsoft.ML.Transforms/Text/CharTokenizeTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -64,12 +64,16 @@ public sealed class Arguments : TransformInputBase
public const string LoaderSignature = "CharToken";
public const string UserName = "Character Tokenizer Transform";

// Keep track of the model that was saved with ver:0x00010001
private readonly bool _isSeparatorStartEnd;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: "CHARTOKN",
verWrittenCur: 0x00010001, // Initial

@TomFinleyTomFinleyJul 21, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

verWrittenCur: 0x00010001, // Initial [](start = 16, length = 37)

See other examples. You will see that we keep commented out the version strings from old versions, to track them and so we know why we bumped the version each time. This is important: we have live code in our deserializers that is meant to handle those version bumps, and for that reason we like to have documentation on what exactly changed, so that when we have tests on this or that version in our deserializers, we know why they changed. #Resolved

verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial
verWrittenCur: 0x00010002, // Updated to use UnitSeparator <US> character instead of using <ETX><STX> for vector inputs.
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}
Expand All@@ -84,6 +88,7 @@ private static VersionInfo GetVersionInfo()
private volatile string _keyValuesStr;
private volatile int[] _keyValuesBoundaries;

private const ushort UnitSeparator = 0x1f;
private const ushort TextStartMarker = 0x02;
private const ushort TextEndMarker = 0x03;
private const int TextMarkersCount = 2;
Expand DownExpand Up@@ -120,6 +125,8 @@ private CharTokenizeTransform(IHost host, ModelLoadContext ctx, IDataView input)
// byte: _useMarkerChars value.
_useMarkerChars = ctx.Reader.ReadBoolByte();

_isSeparatorStartEnd = ctx.Header.ModelVerReadable < 0x00010002 || ctx.Reader.ReadBoolByte();

_type = GetOutputColumnType();
SetMetadata();
}
Expand All@@ -145,6 +152,7 @@ public override void Save(ModelSaveContext ctx)
// byte: _useMarkerChars value.
SaveBase(ctx);
ctx.Writer.WriteBoolByte(_useMarkerChars);
ctx.Writer.WriteBoolByte(_isSeparatorStartEnd);
}

protected override ColumnType GetColumnTypeCore(int iinfo)
Expand DownExpand Up@@ -399,8 +407,8 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

var getSrc = GetSrcGetter<VBuffer<DvText>>(input, iinfo);
var src = default(VBuffer<DvText>);
return
(ref VBuffer<ushort> dst) =>

ValueGetter<VBuffer<ushort>> getterWithStartEndSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

Expand DownExpand Up@@ -438,6 +446,67 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

dst = new VBuffer<ushort>(len, values, dst.Indices);
};

ValueGetter < VBuffer<ushort> > getterWithUnitSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

int len = 0;

for (int i = 0; i < src.Count; i++)
{
if (src.Values[i].HasChars)
{
len += src.Values[i].Length;

if (i > 0)
len += 1; // add UnitSeparator character to len that will be added
}
}

if (_useMarkerChars)
len += TextMarkersCount;

var values = dst.Values;
if (len > 0)
{
if (Utils.Size(values) < len)
values = new ushort[len];

int index = 0;

// VBuffer<DvText> can be a result of either concatenating text columns together
// or application of word tokenizer before char tokenizer in TextTransform.
//
// Considering VBuffer<DvText> as a single text stream.
// Therefore, prepend and append start and end markers only once i.e. at the start and at end of vector.
// Insert UnitSeparator after every piece of text in the vector.
if (_useMarkerChars)
values[index++] = TextStartMarker;

for (int i = 0; i < src.Count; i++)
{
if (!src.Values[i].HasChars)
continue;

if (i > 0)
values[index++] = UnitSeparator;

for (int ich = 0; ich < src.Values[i].Length; ich++)
{
values[index++] = src.Values[i][ich];
}
}

if (_useMarkerChars)
values[index++] = TextEndMarker;

Contracts.Assert(index == len);
}

dst = new VBuffer<ushort>(len, values, dst.Indices);
};
return _isSeparatorStartEnd ? getterWithStartEndSep : getterWithUnitSep;
}
}
}
54 changes: 25 additions & 29 deletions src/Microsoft.ML.Transforms/Text/TextTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -262,6 +262,30 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new ConcatTransform(h, new ConcatTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)

@zeahmedzeahmedJul 19, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@justinormont, I moved TextNormalizer above WordTokenizer. Let me know if there is any adverse effect of it? #Resolved

@justinormontjustinormontJul 20, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't know if there are averse effects of running TextNormalizer before WordTokenizer. I expect it is different, but assume not adversely so. For instance (depending on the specifics of the tokenizer), I don't can be split into {I, do, n't} but if removing punctuation first, it may be split to {I, dont}.

If you make me a build, I'll run a manual regression test. Unfortunately we don't currently have running nightly regression tests to tell us if the text datasets decreased in their core metrics.

#Resolved

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't like this change. It's far from clear to me that if someone asks to remove stopwords that they meant for the content of stopwords to be retained in the chargrams. At the very least this ought to be a configurable option.


In reply to: 204000786 [](ancestors = 204000786)

@zeahmedzeahmedJul 21, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@TomFinley, Can you elaborate more on this? Your point is not clear to me. Don't you see it good to have TextNormalizer applied before WordTokenizer?


In reply to: 204201054 [](ancestors = 204201054,204000786)

@TomFinleyTomFinleyJul 25, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sorry thought you were saying something else for some reason got confused. #Resolved

{
var xfCols = new TextNormalizerCol[textCols.Length];
string[] dstCols = new string[textCols.Length];
for (int i = 0; i < textCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, textCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = textCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

textCols = dstCols;
}

if (tparams.NeedsWordTokenizationTransform)
{
var xfCols = new DelimitedTokenizeTransform.Column[textCols.Length];
Expand All@@ -281,34 +305,6 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new DelimitedTokenizeTransform(h, new DelimitedTokenizeTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)
{
string[] srcCols = wordTokCols == null ? textCols : wordTokCols;
var xfCols = new TextNormalizerCol[srcCols.Length];
string[] dstCols = new string[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, srcCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = srcCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

if (wordTokCols != null)
wordTokCols = dstCols;
else
textCols = dstCols;
}

if (tparams.NeedsRemoveStopwordsTransform)
{
Contracts.Assert(wordTokCols != null, "StopWords transform requires that word tokenization has been applied to the input text.");
Expand DownExpand Up@@ -360,7 +356,7 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
if (tparams.CharExtractorFactory != null)
{
{
var srcCols = wordTokCols ?? textCols;
var srcCols = tparams.NeedsRemoveStopwordsTransform ? wordTokCols : textCols;
charTokCols = new string[srcCols.Length];
var xfCols = new CharTokenizeTransform.Column[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
Expand Down
2 changes: 1 addition & 1 deletion test/Microsoft.ML.Predictor.Tests/TestPipelineSweeper.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -308,7 +308,7 @@ public void PipelineSweeperRoles()
var trainAuc = bestPipeline.PerformanceSummary.TrainingMetricValue;
var testAuc = bestPipeline.PerformanceSummary.MetricValue;
Assert.True((0.94 < trainAuc) && (trainAuc < 0.95));
Assert.True((0.83 < testAuc) && (testAuc < 0.84));
Assert.True((0.815 < testAuc) && (testAuc < 0.825));

var results = runner.GetOutput<IDataView>("ResultsOut");
Assert.NotNull(results);
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
77 changes: 73 additions & 4 deletions src/Microsoft.ML.Transforms/Text/CharTokenizeTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -64,12 +64,16 @@ public sealed class Arguments : TransformInputBase
public const string LoaderSignature = "CharToken";
public const string UserName = "Character Tokenizer Transform";

// Keep track of the model that was saved with ver:0x00010001
private readonly bool _isSeparatorStartEnd;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: "CHARTOKN",
verWrittenCur: 0x00010001, // Initial

@TomFinleyTomFinleyJul 21, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

verWrittenCur: 0x00010001, // Initial [](start = 16, length = 37)

See other examples. You will see that we keep commented out the version strings from old versions, to track them and so we know why we bumped the version each time. This is important: we have live code in our deserializers that is meant to handle those version bumps, and for that reason we like to have documentation on what exactly changed, so that when we have tests on this or that version in our deserializers, we know why they changed. #Resolved

verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial
verWrittenCur: 0x00010002, // Updated to use UnitSeparator <US> character instead of using <ETX><STX> for vector inputs.
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}
Expand All@@ -84,6 +88,7 @@ private static VersionInfo GetVersionInfo()
private volatile string _keyValuesStr;
private volatile int[] _keyValuesBoundaries;

private const ushort UnitSeparator = 0x1f;
private const ushort TextStartMarker = 0x02;
private const ushort TextEndMarker = 0x03;
private const int TextMarkersCount = 2;
Expand DownExpand Up@@ -120,6 +125,8 @@ private CharTokenizeTransform(IHost host, ModelLoadContext ctx, IDataView input)
// byte: _useMarkerChars value.
_useMarkerChars = ctx.Reader.ReadBoolByte();

_isSeparatorStartEnd = ctx.Header.ModelVerReadable < 0x00010002 || ctx.Reader.ReadBoolByte();

_type = GetOutputColumnType();
SetMetadata();
}
Expand All@@ -145,6 +152,7 @@ public override void Save(ModelSaveContext ctx)
// byte: _useMarkerChars value.
SaveBase(ctx);
ctx.Writer.WriteBoolByte(_useMarkerChars);
ctx.Writer.WriteBoolByte(_isSeparatorStartEnd);
}

protected override ColumnType GetColumnTypeCore(int iinfo)
Expand DownExpand Up@@ -399,8 +407,8 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

var getSrc = GetSrcGetter<VBuffer<DvText>>(input, iinfo);
var src = default(VBuffer<DvText>);
return
(ref VBuffer<ushort> dst) =>

ValueGetter<VBuffer<ushort>> getterWithStartEndSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

Expand DownExpand Up@@ -438,6 +446,67 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

dst = new VBuffer<ushort>(len, values, dst.Indices);
};

ValueGetter < VBuffer<ushort> > getterWithUnitSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

int len = 0;

for (int i = 0; i < src.Count; i++)
{
if (src.Values[i].HasChars)
{
len += src.Values[i].Length;

if (i > 0)
len += 1; // add UnitSeparator character to len that will be added
}
}

if (_useMarkerChars)
len += TextMarkersCount;

var values = dst.Values;
if (len > 0)
{
if (Utils.Size(values) < len)
values = new ushort[len];

int index = 0;

// VBuffer<DvText> can be a result of either concatenating text columns together
// or application of word tokenizer before char tokenizer in TextTransform.
//
// Considering VBuffer<DvText> as a single text stream.
// Therefore, prepend and append start and end markers only once i.e. at the start and at end of vector.
// Insert UnitSeparator after every piece of text in the vector.
if (_useMarkerChars)
values[index++] = TextStartMarker;

for (int i = 0; i < src.Count; i++)
{
if (!src.Values[i].HasChars)
continue;

if (i > 0)
values[index++] = UnitSeparator;

for (int ich = 0; ich < src.Values[i].Length; ich++)
{
values[index++] = src.Values[i][ich];
}
}

if (_useMarkerChars)
values[index++] = TextEndMarker;

Contracts.Assert(index == len);
}

dst = new VBuffer<ushort>(len, values, dst.Indices);
};
return _isSeparatorStartEnd ? getterWithStartEndSep : getterWithUnitSep;
}
}
}
54 changes: 25 additions & 29 deletions src/Microsoft.ML.Transforms/Text/TextTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -262,6 +262,30 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new ConcatTransform(h, new ConcatTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)

@zeahmedzeahmedJul 19, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@justinormont, I moved TextNormalizer above WordTokenizer. Let me know if there is any adverse effect of it? #Resolved

@justinormontjustinormontJul 20, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't know if there are averse effects of running TextNormalizer before WordTokenizer. I expect it is different, but assume not adversely so. For instance (depending on the specifics of the tokenizer), I don't can be split into {I, do, n't} but if removing punctuation first, it may be split to {I, dont}.

If you make me a build, I'll run a manual regression test. Unfortunately we don't currently have running nightly regression tests to tell us if the text datasets decreased in their core metrics.

#Resolved

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't like this change. It's far from clear to me that if someone asks to remove stopwords that they meant for the content of stopwords to be retained in the chargrams. At the very least this ought to be a configurable option.


In reply to: 204000786 [](ancestors = 204000786)

@zeahmedzeahmedJul 21, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@TomFinley, Can you elaborate more on this? Your point is not clear to me. Don't you see it good to have TextNormalizer applied before WordTokenizer?


In reply to: 204201054 [](ancestors = 204201054,204000786)

@TomFinleyTomFinleyJul 25, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sorry thought you were saying something else for some reason got confused. #Resolved

{
var xfCols = new TextNormalizerCol[textCols.Length];
string[] dstCols = new string[textCols.Length];
for (int i = 0; i < textCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, textCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = textCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

textCols = dstCols;
}

if (tparams.NeedsWordTokenizationTransform)
{
var xfCols = new DelimitedTokenizeTransform.Column[textCols.Length];
Expand All@@ -281,34 +305,6 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new DelimitedTokenizeTransform(h, new DelimitedTokenizeTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)
{
string[] srcCols = wordTokCols == null ? textCols : wordTokCols;
var xfCols = new TextNormalizerCol[srcCols.Length];
string[] dstCols = new string[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, srcCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = srcCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

if (wordTokCols != null)
wordTokCols = dstCols;
else
textCols = dstCols;
}

if (tparams.NeedsRemoveStopwordsTransform)
{
Contracts.Assert(wordTokCols != null, "StopWords transform requires that word tokenization has been applied to the input text.");
Expand DownExpand Up@@ -360,7 +356,7 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
if (tparams.CharExtractorFactory != null)
{
{
var srcCols = wordTokCols ?? textCols;
var srcCols = tparams.NeedsRemoveStopwordsTransform ? wordTokCols : textCols;
charTokCols = new string[srcCols.Length];
var xfCols = new CharTokenizeTransform.Column[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
Expand Down
2 changes: 1 addition & 1 deletion test/Microsoft.ML.Predictor.Tests/TestPipelineSweeper.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -308,7 +308,7 @@ public void PipelineSweeperRoles()
var trainAuc = bestPipeline.PerformanceSummary.TrainingMetricValue;
var testAuc = bestPipeline.PerformanceSummary.MetricValue;
Assert.True((0.94 < trainAuc) && (trainAuc < 0.95));
Assert.True((0.83 < testAuc) && (testAuc < 0.84));
Assert.True((0.815 < testAuc) && (testAuc < 0.825));

var results = runner.GetOutput<IDataView>("ResultsOut");
Assert.NotNull(results);
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
77 changes: 73 additions & 4 deletions src/Microsoft.ML.Transforms/Text/CharTokenizeTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -64,12 +64,16 @@ public sealed class Arguments : TransformInputBase
public const string LoaderSignature = "CharToken";
public const string UserName = "Character Tokenizer Transform";

// Keep track of the model that was saved with ver:0x00010001
private readonly bool _isSeparatorStartEnd;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: "CHARTOKN",
verWrittenCur: 0x00010001, // Initial

@TomFinleyTomFinleyJul 21, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

verWrittenCur: 0x00010001, // Initial [](start = 16, length = 37)

See other examples. You will see that we keep commented out the version strings from old versions, to track them and so we know why we bumped the version each time. This is important: we have live code in our deserializers that is meant to handle those version bumps, and for that reason we like to have documentation on what exactly changed, so that when we have tests on this or that version in our deserializers, we know why they changed. #Resolved

verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial
verWrittenCur: 0x00010002, // Updated to use UnitSeparator <US> character instead of using <ETX><STX> for vector inputs.
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}
Expand All@@ -84,6 +88,7 @@ private static VersionInfo GetVersionInfo()
private volatile string _keyValuesStr;
private volatile int[] _keyValuesBoundaries;

private const ushort UnitSeparator = 0x1f;
private const ushort TextStartMarker = 0x02;
private const ushort TextEndMarker = 0x03;
private const int TextMarkersCount = 2;
Expand DownExpand Up@@ -120,6 +125,8 @@ private CharTokenizeTransform(IHost host, ModelLoadContext ctx, IDataView input)
// byte: _useMarkerChars value.
_useMarkerChars = ctx.Reader.ReadBoolByte();

_isSeparatorStartEnd = ctx.Header.ModelVerReadable < 0x00010002 || ctx.Reader.ReadBoolByte();

_type = GetOutputColumnType();
SetMetadata();
}
Expand All@@ -145,6 +152,7 @@ public override void Save(ModelSaveContext ctx)
// byte: _useMarkerChars value.
SaveBase(ctx);
ctx.Writer.WriteBoolByte(_useMarkerChars);
ctx.Writer.WriteBoolByte(_isSeparatorStartEnd);
}

protected override ColumnType GetColumnTypeCore(int iinfo)
Expand DownExpand Up@@ -399,8 +407,8 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

var getSrc = GetSrcGetter<VBuffer<DvText>>(input, iinfo);
var src = default(VBuffer<DvText>);
return
(ref VBuffer<ushort> dst) =>

ValueGetter<VBuffer<ushort>> getterWithStartEndSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

Expand DownExpand Up@@ -438,6 +446,67 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

dst = new VBuffer<ushort>(len, values, dst.Indices);
};

ValueGetter < VBuffer<ushort> > getterWithUnitSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

int len = 0;

for (int i = 0; i < src.Count; i++)
{
if (src.Values[i].HasChars)
{
len += src.Values[i].Length;

if (i > 0)
len += 1; // add UnitSeparator character to len that will be added
}
}

if (_useMarkerChars)
len += TextMarkersCount;

var values = dst.Values;
if (len > 0)
{
if (Utils.Size(values) < len)
values = new ushort[len];

int index = 0;

// VBuffer<DvText> can be a result of either concatenating text columns together
// or application of word tokenizer before char tokenizer in TextTransform.
//
// Considering VBuffer<DvText> as a single text stream.
// Therefore, prepend and append start and end markers only once i.e. at the start and at end of vector.
// Insert UnitSeparator after every piece of text in the vector.
if (_useMarkerChars)
values[index++] = TextStartMarker;

for (int i = 0; i < src.Count; i++)
{
if (!src.Values[i].HasChars)
continue;

if (i > 0)
values[index++] = UnitSeparator;

for (int ich = 0; ich < src.Values[i].Length; ich++)
{
values[index++] = src.Values[i][ich];
}
}

if (_useMarkerChars)
values[index++] = TextEndMarker;

Contracts.Assert(index == len);
}

dst = new VBuffer<ushort>(len, values, dst.Indices);
};
return _isSeparatorStartEnd ? getterWithStartEndSep : getterWithUnitSep;
}
}
}
54 changes: 25 additions & 29 deletions src/Microsoft.ML.Transforms/Text/TextTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -262,6 +262,30 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new ConcatTransform(h, new ConcatTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)

@zeahmedzeahmedJul 19, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@justinormont, I moved TextNormalizer above WordTokenizer. Let me know if there is any adverse effect of it? #Resolved

@justinormontjustinormontJul 20, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't know if there are averse effects of running TextNormalizer before WordTokenizer. I expect it is different, but assume not adversely so. For instance (depending on the specifics of the tokenizer), I don't can be split into {I, do, n't} but if removing punctuation first, it may be split to {I, dont}.

If you make me a build, I'll run a manual regression test. Unfortunately we don't currently have running nightly regression tests to tell us if the text datasets decreased in their core metrics.

#Resolved

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't like this change. It's far from clear to me that if someone asks to remove stopwords that they meant for the content of stopwords to be retained in the chargrams. At the very least this ought to be a configurable option.


In reply to: 204000786 [](ancestors = 204000786)

@zeahmedzeahmedJul 21, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@TomFinley, Can you elaborate more on this? Your point is not clear to me. Don't you see it good to have TextNormalizer applied before WordTokenizer?


In reply to: 204201054 [](ancestors = 204201054,204000786)

@TomFinleyTomFinleyJul 25, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sorry thought you were saying something else for some reason got confused. #Resolved

{
var xfCols = new TextNormalizerCol[textCols.Length];
string[] dstCols = new string[textCols.Length];
for (int i = 0; i < textCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, textCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = textCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

textCols = dstCols;
}

if (tparams.NeedsWordTokenizationTransform)
{
var xfCols = new DelimitedTokenizeTransform.Column[textCols.Length];
Expand All@@ -281,34 +305,6 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new DelimitedTokenizeTransform(h, new DelimitedTokenizeTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)
{
string[] srcCols = wordTokCols == null ? textCols : wordTokCols;
var xfCols = new TextNormalizerCol[srcCols.Length];
string[] dstCols = new string[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, srcCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = srcCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

if (wordTokCols != null)
wordTokCols = dstCols;
else
textCols = dstCols;
}

if (tparams.NeedsRemoveStopwordsTransform)
{
Contracts.Assert(wordTokCols != null, "StopWords transform requires that word tokenization has been applied to the input text.");
Expand DownExpand Up@@ -360,7 +356,7 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
if (tparams.CharExtractorFactory != null)
{
{
var srcCols = wordTokCols ?? textCols;
var srcCols = tparams.NeedsRemoveStopwordsTransform ? wordTokCols : textCols;
charTokCols = new string[srcCols.Length];
var xfCols = new CharTokenizeTransform.Column[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
Expand Down
2 changes: 1 addition & 1 deletion test/Microsoft.ML.Predictor.Tests/TestPipelineSweeper.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -308,7 +308,7 @@ public void PipelineSweeperRoles()
var trainAuc = bestPipeline.PerformanceSummary.TrainingMetricValue;
var testAuc = bestPipeline.PerformanceSummary.MetricValue;
Assert.True((0.94 < trainAuc) && (trainAuc < 0.95));
Assert.True((0.83 < testAuc) && (testAuc < 0.84));
Assert.True((0.815 < testAuc) && (testAuc < 0.825));

var results = runner.GetOutput<IDataView>("ResultsOut");
Assert.NotNull(results);
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
77 changes: 73 additions & 4 deletions src/Microsoft.ML.Transforms/Text/CharTokenizeTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -64,12 +64,16 @@ public sealed class Arguments : TransformInputBase
public const string LoaderSignature = "CharToken";
public const string UserName = "Character Tokenizer Transform";

// Keep track of the model that was saved with ver:0x00010001
private readonly bool _isSeparatorStartEnd;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: "CHARTOKN",
verWrittenCur: 0x00010001, // Initial

@TomFinleyTomFinleyJul 21, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

verWrittenCur: 0x00010001, // Initial [](start = 16, length = 37)

See other examples. You will see that we keep commented out the version strings from old versions, to track them and so we know why we bumped the version each time. This is important: we have live code in our deserializers that is meant to handle those version bumps, and for that reason we like to have documentation on what exactly changed, so that when we have tests on this or that version in our deserializers, we know why they changed. #Resolved

verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial
verWrittenCur: 0x00010002, // Updated to use UnitSeparator <US> character instead of using <ETX><STX> for vector inputs.
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}
Expand All@@ -84,6 +88,7 @@ private static VersionInfo GetVersionInfo()
private volatile string _keyValuesStr;
private volatile int[] _keyValuesBoundaries;

private const ushort UnitSeparator = 0x1f;
private const ushort TextStartMarker = 0x02;
private const ushort TextEndMarker = 0x03;
private const int TextMarkersCount = 2;
Expand DownExpand Up@@ -120,6 +125,8 @@ private CharTokenizeTransform(IHost host, ModelLoadContext ctx, IDataView input)
// byte: _useMarkerChars value.
_useMarkerChars = ctx.Reader.ReadBoolByte();

_isSeparatorStartEnd = ctx.Header.ModelVerReadable < 0x00010002 || ctx.Reader.ReadBoolByte();

_type = GetOutputColumnType();
SetMetadata();
}
Expand All@@ -145,6 +152,7 @@ public override void Save(ModelSaveContext ctx)
// byte: _useMarkerChars value.
SaveBase(ctx);
ctx.Writer.WriteBoolByte(_useMarkerChars);
ctx.Writer.WriteBoolByte(_isSeparatorStartEnd);
}

protected override ColumnType GetColumnTypeCore(int iinfo)
Expand DownExpand Up@@ -399,8 +407,8 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

var getSrc = GetSrcGetter<VBuffer<DvText>>(input, iinfo);
var src = default(VBuffer<DvText>);
return
(ref VBuffer<ushort> dst) =>

ValueGetter<VBuffer<ushort>> getterWithStartEndSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

Expand DownExpand Up@@ -438,6 +446,67 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

dst = new VBuffer<ushort>(len, values, dst.Indices);
};

ValueGetter < VBuffer<ushort> > getterWithUnitSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

int len = 0;

for (int i = 0; i < src.Count; i++)
{
if (src.Values[i].HasChars)
{
len += src.Values[i].Length;

if (i > 0)
len += 1; // add UnitSeparator character to len that will be added
}
}

if (_useMarkerChars)
len += TextMarkersCount;

var values = dst.Values;
if (len > 0)
{
if (Utils.Size(values) < len)
values = new ushort[len];

int index = 0;

// VBuffer<DvText> can be a result of either concatenating text columns together
// or application of word tokenizer before char tokenizer in TextTransform.
//
// Considering VBuffer<DvText> as a single text stream.
// Therefore, prepend and append start and end markers only once i.e. at the start and at end of vector.
// Insert UnitSeparator after every piece of text in the vector.
if (_useMarkerChars)
values[index++] = TextStartMarker;

for (int i = 0; i < src.Count; i++)
{
if (!src.Values[i].HasChars)
continue;

if (i > 0)
values[index++] = UnitSeparator;

for (int ich = 0; ich < src.Values[i].Length; ich++)
{
values[index++] = src.Values[i][ich];
}
}

if (_useMarkerChars)
values[index++] = TextEndMarker;

Contracts.Assert(index == len);
}

dst = new VBuffer<ushort>(len, values, dst.Indices);
};
return _isSeparatorStartEnd ? getterWithStartEndSep : getterWithUnitSep;
}
}
}
54 changes: 25 additions & 29 deletions src/Microsoft.ML.Transforms/Text/TextTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -262,6 +262,30 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new ConcatTransform(h, new ConcatTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)

@zeahmedzeahmedJul 19, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@justinormont, I moved TextNormalizer above WordTokenizer. Let me know if there is any adverse effect of it? #Resolved

@justinormontjustinormontJul 20, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't know if there are averse effects of running TextNormalizer before WordTokenizer. I expect it is different, but assume not adversely so. For instance (depending on the specifics of the tokenizer), I don't can be split into {I, do, n't} but if removing punctuation first, it may be split to {I, dont}.

If you make me a build, I'll run a manual regression test. Unfortunately we don't currently have running nightly regression tests to tell us if the text datasets decreased in their core metrics.

#Resolved

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't like this change. It's far from clear to me that if someone asks to remove stopwords that they meant for the content of stopwords to be retained in the chargrams. At the very least this ought to be a configurable option.


In reply to: 204000786 [](ancestors = 204000786)

@zeahmedzeahmedJul 21, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@TomFinley, Can you elaborate more on this? Your point is not clear to me. Don't you see it good to have TextNormalizer applied before WordTokenizer?


In reply to: 204201054 [](ancestors = 204201054,204000786)

@TomFinleyTomFinleyJul 25, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sorry thought you were saying something else for some reason got confused. #Resolved

{
var xfCols = new TextNormalizerCol[textCols.Length];
string[] dstCols = new string[textCols.Length];
for (int i = 0; i < textCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, textCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = textCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

textCols = dstCols;
}

if (tparams.NeedsWordTokenizationTransform)
{
var xfCols = new DelimitedTokenizeTransform.Column[textCols.Length];
Expand All@@ -281,34 +305,6 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new DelimitedTokenizeTransform(h, new DelimitedTokenizeTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)
{
string[] srcCols = wordTokCols == null ? textCols : wordTokCols;
var xfCols = new TextNormalizerCol[srcCols.Length];
string[] dstCols = new string[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, srcCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = srcCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

if (wordTokCols != null)
wordTokCols = dstCols;
else
textCols = dstCols;
}

if (tparams.NeedsRemoveStopwordsTransform)
{
Contracts.Assert(wordTokCols != null, "StopWords transform requires that word tokenization has been applied to the input text.");
Expand DownExpand Up@@ -360,7 +356,7 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
if (tparams.CharExtractorFactory != null)
{
{
var srcCols = wordTokCols ?? textCols;
var srcCols = tparams.NeedsRemoveStopwordsTransform ? wordTokCols : textCols;
charTokCols = new string[srcCols.Length];
var xfCols = new CharTokenizeTransform.Column[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
Expand Down
2 changes: 1 addition & 1 deletion test/Microsoft.ML.Predictor.Tests/TestPipelineSweeper.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -308,7 +308,7 @@ public void PipelineSweeperRoles()
var trainAuc = bestPipeline.PerformanceSummary.TrainingMetricValue;
var testAuc = bestPipeline.PerformanceSummary.MetricValue;
Assert.True((0.94 < trainAuc) && (trainAuc < 0.95));
Assert.True((0.83 < testAuc) && (testAuc < 0.84));
Assert.True((0.815 < testAuc) && (testAuc < 0.825));

var results = runner.GetOutput<IDataView>("ResultsOut");
Assert.NotNull(results);
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
77 changes: 73 additions & 4 deletions src/Microsoft.ML.Transforms/Text/CharTokenizeTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -64,12 +64,16 @@ public sealed class Arguments : TransformInputBase
public const string LoaderSignature = "CharToken";
public const string UserName = "Character Tokenizer Transform";

// Keep track of the model that was saved with ver:0x00010001
private readonly bool _isSeparatorStartEnd;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: "CHARTOKN",
verWrittenCur: 0x00010001, // Initial

@TomFinleyTomFinleyJul 21, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

verWrittenCur: 0x00010001, // Initial [](start = 16, length = 37)

See other examples. You will see that we keep commented out the version strings from old versions, to track them and so we know why we bumped the version each time. This is important: we have live code in our deserializers that is meant to handle those version bumps, and for that reason we like to have documentation on what exactly changed, so that when we have tests on this or that version in our deserializers, we know why they changed. #Resolved

verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial
verWrittenCur: 0x00010002, // Updated to use UnitSeparator <US> character instead of using <ETX><STX> for vector inputs.
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}
Expand All@@ -84,6 +88,7 @@ private static VersionInfo GetVersionInfo()
private volatile string _keyValuesStr;
private volatile int[] _keyValuesBoundaries;

private const ushort UnitSeparator = 0x1f;
private const ushort TextStartMarker = 0x02;
private const ushort TextEndMarker = 0x03;
private const int TextMarkersCount = 2;
Expand DownExpand Up@@ -120,6 +125,8 @@ private CharTokenizeTransform(IHost host, ModelLoadContext ctx, IDataView input)
// byte: _useMarkerChars value.
_useMarkerChars = ctx.Reader.ReadBoolByte();

_isSeparatorStartEnd = ctx.Header.ModelVerReadable < 0x00010002 || ctx.Reader.ReadBoolByte();

_type = GetOutputColumnType();
SetMetadata();
}
Expand All@@ -145,6 +152,7 @@ public override void Save(ModelSaveContext ctx)
// byte: _useMarkerChars value.
SaveBase(ctx);
ctx.Writer.WriteBoolByte(_useMarkerChars);
ctx.Writer.WriteBoolByte(_isSeparatorStartEnd);
}

protected override ColumnType GetColumnTypeCore(int iinfo)
Expand DownExpand Up@@ -399,8 +407,8 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

var getSrc = GetSrcGetter<VBuffer<DvText>>(input, iinfo);
var src = default(VBuffer<DvText>);
return
(ref VBuffer<ushort> dst) =>

ValueGetter<VBuffer<ushort>> getterWithStartEndSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

Expand DownExpand Up@@ -438,6 +446,67 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

dst = new VBuffer<ushort>(len, values, dst.Indices);
};

ValueGetter < VBuffer<ushort> > getterWithUnitSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

int len = 0;

for (int i = 0; i < src.Count; i++)
{
if (src.Values[i].HasChars)
{
len += src.Values[i].Length;

if (i > 0)
len += 1; // add UnitSeparator character to len that will be added
}
}

if (_useMarkerChars)
len += TextMarkersCount;

var values = dst.Values;
if (len > 0)
{
if (Utils.Size(values) < len)
values = new ushort[len];

int index = 0;

// VBuffer<DvText> can be a result of either concatenating text columns together
// or application of word tokenizer before char tokenizer in TextTransform.
//
// Considering VBuffer<DvText> as a single text stream.
// Therefore, prepend and append start and end markers only once i.e. at the start and at end of vector.
// Insert UnitSeparator after every piece of text in the vector.
if (_useMarkerChars)
values[index++] = TextStartMarker;

for (int i = 0; i < src.Count; i++)
{
if (!src.Values[i].HasChars)
continue;

if (i > 0)
values[index++] = UnitSeparator;

for (int ich = 0; ich < src.Values[i].Length; ich++)
{
values[index++] = src.Values[i][ich];
}
}

if (_useMarkerChars)
values[index++] = TextEndMarker;

Contracts.Assert(index == len);
}

dst = new VBuffer<ushort>(len, values, dst.Indices);
};
return _isSeparatorStartEnd ? getterWithStartEndSep : getterWithUnitSep;
}
}
}
54 changes: 25 additions & 29 deletions src/Microsoft.ML.Transforms/Text/TextTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -262,6 +262,30 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new ConcatTransform(h, new ConcatTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)

@zeahmedzeahmedJul 19, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@justinormont, I moved TextNormalizer above WordTokenizer. Let me know if there is any adverse effect of it? #Resolved

@justinormontjustinormontJul 20, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't know if there are averse effects of running TextNormalizer before WordTokenizer. I expect it is different, but assume not adversely so. For instance (depending on the specifics of the tokenizer), I don't can be split into {I, do, n't} but if removing punctuation first, it may be split to {I, dont}.

If you make me a build, I'll run a manual regression test. Unfortunately we don't currently have running nightly regression tests to tell us if the text datasets decreased in their core metrics.

#Resolved

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't like this change. It's far from clear to me that if someone asks to remove stopwords that they meant for the content of stopwords to be retained in the chargrams. At the very least this ought to be a configurable option.


In reply to: 204000786 [](ancestors = 204000786)

@zeahmedzeahmedJul 21, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@TomFinley, Can you elaborate more on this? Your point is not clear to me. Don't you see it good to have TextNormalizer applied before WordTokenizer?


In reply to: 204201054 [](ancestors = 204201054,204000786)

@TomFinleyTomFinleyJul 25, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sorry thought you were saying something else for some reason got confused. #Resolved

{
var xfCols = new TextNormalizerCol[textCols.Length];
string[] dstCols = new string[textCols.Length];
for (int i = 0; i < textCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, textCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = textCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

textCols = dstCols;
}

if (tparams.NeedsWordTokenizationTransform)
{
var xfCols = new DelimitedTokenizeTransform.Column[textCols.Length];
Expand All@@ -281,34 +305,6 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new DelimitedTokenizeTransform(h, new DelimitedTokenizeTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)
{
string[] srcCols = wordTokCols == null ? textCols : wordTokCols;
var xfCols = new TextNormalizerCol[srcCols.Length];
string[] dstCols = new string[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, srcCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = srcCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

if (wordTokCols != null)
wordTokCols = dstCols;
else
textCols = dstCols;
}

if (tparams.NeedsRemoveStopwordsTransform)
{
Contracts.Assert(wordTokCols != null, "StopWords transform requires that word tokenization has been applied to the input text.");
Expand DownExpand Up@@ -360,7 +356,7 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
if (tparams.CharExtractorFactory != null)
{
{
var srcCols = wordTokCols ?? textCols;
var srcCols = tparams.NeedsRemoveStopwordsTransform ? wordTokCols : textCols;
charTokCols = new string[srcCols.Length];
var xfCols = new CharTokenizeTransform.Column[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
Expand Down
2 changes: 1 addition & 1 deletion test/Microsoft.ML.Predictor.Tests/TestPipelineSweeper.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -308,7 +308,7 @@ public void PipelineSweeperRoles()
var trainAuc = bestPipeline.PerformanceSummary.TrainingMetricValue;
var testAuc = bestPipeline.PerformanceSummary.MetricValue;
Assert.True((0.94 < trainAuc) && (trainAuc < 0.95));
Assert.True((0.83 < testAuc) && (testAuc < 0.84));
Assert.True((0.815 < testAuc) && (testAuc < 0.825));

var results = runner.GetOutput<IDataView>("ResultsOut");
Assert.NotNull(results);
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
77 changes: 73 additions & 4 deletions src/Microsoft.ML.Transforms/Text/CharTokenizeTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -64,12 +64,16 @@ public sealed class Arguments : TransformInputBase
public const string LoaderSignature = "CharToken";
public const string UserName = "Character Tokenizer Transform";

// Keep track of the model that was saved with ver:0x00010001
private readonly bool _isSeparatorStartEnd;

private static VersionInfo GetVersionInfo()
{
return new VersionInfo(
modelSignature: "CHARTOKN",
verWrittenCur: 0x00010001, // Initial

@TomFinleyTomFinleyJul 21, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

verWrittenCur: 0x00010001, // Initial [](start = 16, length = 37)

See other examples. You will see that we keep commented out the version strings from old versions, to track them and so we know why we bumped the version each time. This is important: we have live code in our deserializers that is meant to handle those version bumps, and for that reason we like to have documentation on what exactly changed, so that when we have tests on this or that version in our deserializers, we know why they changed. #Resolved

verReadableCur: 0x00010001,
//verWrittenCur: 0x00010001, // Initial
verWrittenCur: 0x00010002, // Updated to use UnitSeparator <US> character instead of using <ETX><STX> for vector inputs.
verReadableCur: 0x00010002,
verWeCanReadBack: 0x00010001,
loaderSignature: LoaderSignature);
}
Expand All@@ -84,6 +88,7 @@ private static VersionInfo GetVersionInfo()
private volatile string _keyValuesStr;
private volatile int[] _keyValuesBoundaries;

private const ushort UnitSeparator = 0x1f;
private const ushort TextStartMarker = 0x02;
private const ushort TextEndMarker = 0x03;
private const int TextMarkersCount = 2;
Expand DownExpand Up@@ -120,6 +125,8 @@ private CharTokenizeTransform(IHost host, ModelLoadContext ctx, IDataView input)
// byte: _useMarkerChars value.
_useMarkerChars = ctx.Reader.ReadBoolByte();

_isSeparatorStartEnd = ctx.Header.ModelVerReadable < 0x00010002 || ctx.Reader.ReadBoolByte();

_type = GetOutputColumnType();
SetMetadata();
}
Expand All@@ -145,6 +152,7 @@ public override void Save(ModelSaveContext ctx)
// byte: _useMarkerChars value.
SaveBase(ctx);
ctx.Writer.WriteBoolByte(_useMarkerChars);
ctx.Writer.WriteBoolByte(_isSeparatorStartEnd);
}

protected override ColumnType GetColumnTypeCore(int iinfo)
Expand DownExpand Up@@ -399,8 +407,8 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

var getSrc = GetSrcGetter<VBuffer<DvText>>(input, iinfo);
var src = default(VBuffer<DvText>);
return
(ref VBuffer<ushort> dst) =>

ValueGetter<VBuffer<ushort>> getterWithStartEndSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

Expand DownExpand Up@@ -438,6 +446,67 @@ private ValueGetter<VBuffer<ushort>> MakeGetterVec(IRow input, int iinfo)

dst = new VBuffer<ushort>(len, values, dst.Indices);
};

ValueGetter < VBuffer<ushort> > getterWithUnitSep = (ref VBuffer<ushort> dst) =>
{
getSrc(ref src);

int len = 0;

for (int i = 0; i < src.Count; i++)
{
if (src.Values[i].HasChars)
{
len += src.Values[i].Length;

if (i > 0)
len += 1; // add UnitSeparator character to len that will be added
}
}

if (_useMarkerChars)
len += TextMarkersCount;

var values = dst.Values;
if (len > 0)
{
if (Utils.Size(values) < len)
values = new ushort[len];

int index = 0;

// VBuffer<DvText> can be a result of either concatenating text columns together
// or application of word tokenizer before char tokenizer in TextTransform.
//
// Considering VBuffer<DvText> as a single text stream.
// Therefore, prepend and append start and end markers only once i.e. at the start and at end of vector.
// Insert UnitSeparator after every piece of text in the vector.
if (_useMarkerChars)
values[index++] = TextStartMarker;

for (int i = 0; i < src.Count; i++)
{
if (!src.Values[i].HasChars)
continue;

if (i > 0)
values[index++] = UnitSeparator;

for (int ich = 0; ich < src.Values[i].Length; ich++)
{
values[index++] = src.Values[i][ich];
}
}

if (_useMarkerChars)
values[index++] = TextEndMarker;

Contracts.Assert(index == len);
}

dst = new VBuffer<ushort>(len, values, dst.Indices);
};
return _isSeparatorStartEnd ? getterWithStartEndSep : getterWithUnitSep;
}
}
}
54 changes: 25 additions & 29 deletions src/Microsoft.ML.Transforms/Text/TextTransform.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -262,6 +262,30 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new ConcatTransform(h, new ConcatTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)

@zeahmedzeahmedJul 19, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@justinormont, I moved TextNormalizer above WordTokenizer. Let me know if there is any adverse effect of it? #Resolved

@justinormontjustinormontJul 20, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't know if there are averse effects of running TextNormalizer before WordTokenizer. I expect it is different, but assume not adversely so. For instance (depending on the specifics of the tokenizer), I don't can be split into {I, do, n't} but if removing punctuation first, it may be split to {I, dont}.

If you make me a build, I'll run a manual regression test. Unfortunately we don't currently have running nightly regression tests to tell us if the text datasets decreased in their core metrics.

#Resolved

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't like this change. It's far from clear to me that if someone asks to remove stopwords that they meant for the content of stopwords to be retained in the chargrams. At the very least this ought to be a configurable option.


In reply to: 204000786 [](ancestors = 204000786)

@zeahmedzeahmedJul 21, 2018

Copy link
Copy Markdown
ContributorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@TomFinley, Can you elaborate more on this? Your point is not clear to me. Don't you see it good to have TextNormalizer applied before WordTokenizer?


In reply to: 204201054 [](ancestors = 204201054,204000786)

@TomFinleyTomFinleyJul 25, 2018

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sorry thought you were saying something else for some reason got confused. #Resolved

{
var xfCols = new TextNormalizerCol[textCols.Length];
string[] dstCols = new string[textCols.Length];
for (int i = 0; i < textCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, textCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = textCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

textCols = dstCols;
}

if (tparams.NeedsWordTokenizationTransform)
{
var xfCols = new DelimitedTokenizeTransform.Column[textCols.Length];
Expand All@@ -281,34 +305,6 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
view = new DelimitedTokenizeTransform(h, new DelimitedTokenizeTransform.Arguments() { Column = xfCols }, view);
}

if (tparams.NeedsNormalizeTransform)
{
string[] srcCols = wordTokCols == null ? textCols : wordTokCols;
var xfCols = new TextNormalizerCol[srcCols.Length];
string[] dstCols = new string[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
{
dstCols[i] = GenerateColumnName(view.Schema, srcCols[i], "TextNormalizer");
tempCols.Add(dstCols[i]);
xfCols[i] = new TextNormalizerCol() { Source = srcCols[i], Name = dstCols[i] };
}

view = new TextNormalizerTransform(h,
new TextNormalizerArgs()
{
Column = xfCols,
KeepDiacritics = tparams.KeepDiacritics,
KeepNumbers = tparams.KeepNumbers,
KeepPunctuations = tparams.KeepPunctuations,
TextCase = tparams.TextCase
}, view);

if (wordTokCols != null)
wordTokCols = dstCols;
else
textCols = dstCols;
}

if (tparams.NeedsRemoveStopwordsTransform)
{
Contracts.Assert(wordTokCols != null, "StopWords transform requires that word tokenization has been applied to the input text.");
Expand DownExpand Up@@ -360,7 +356,7 @@ public static IDataTransform Create(IHostEnvironment env, Arguments args, IDataV
if (tparams.CharExtractorFactory != null)
{
{
var srcCols = wordTokCols ?? textCols;
var srcCols = tparams.NeedsRemoveStopwordsTransform ? wordTokCols : textCols;
charTokCols = new string[srcCols.Length];
var xfCols = new CharTokenizeTransform.Column[srcCols.Length];
for (int i = 0; i < srcCols.Length; i++)
Expand Down
2 changes: 1 addition & 1 deletion test/Microsoft.ML.Predictor.Tests/TestPipelineSweeper.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -308,7 +308,7 @@ public void PipelineSweeperRoles()
var trainAuc = bestPipeline.PerformanceSummary.TrainingMetricValue;
var testAuc = bestPipeline.PerformanceSummary.MetricValue;
Assert.True((0.94 < trainAuc) && (trainAuc < 0.95));
Assert.True((0.83 < testAuc) && (testAuc < 0.84));
Assert.True((0.815 < testAuc) && (testAuc < 0.825));

var results = runner.GetOutput<IDataView>("ResultsOut");
Assert.NotNull(results);
Expand Down
Loading