Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
148 changes: 130 additions & 18 deletions src/Microsoft.ML.AutoML/API/AutoCatalog.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,8 +4,11 @@

using System;
using System.Collections.Generic;
using System.Diagnostics.Contracts;
using System.Linq;
using Microsoft.ML.AutoML.CodeGen;
using Microsoft.ML.Data;
using Microsoft.ML.Runtime;
using Microsoft.ML.SearchSpace;
using Microsoft.ML.Trainers.FastTree;

Expand DownExpand Up@@ -538,55 +541,164 @@ public SweepableEstimator[] Regression(string labelColumnName = DefaultColumnNam
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] TextFeaturizer(string outputColumnName, string inputColumnName)
{
throw new NotImplementedException();
var option = new FeaturizeTextOption
{
InputColumnName = inputColumnName,
OutputColumnName = outputColumnName,
};

return new[] { SweepableEstimatorFactory.CreateFeaturizeText(option) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing numeric columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] NumericFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] NumericFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.CheckValue(inputColumnNames, nameof(inputColumnNames));
Contracts.CheckValue(outputColumnNames, nameof(outputColumnNames));
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this will blow up if either of those inputs are null. Make sure they aren't null first.

var replaceMissingValueOption = new ReplaceMissingValueOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new[] { SweepableEstimatorFactory.CreateReplaceMissingValues(replaceMissingValueOption) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing catalog columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] CatalogFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] CatalogFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

var option = new OneHotOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new SweepableEstimator[] { SweepableEstimatorFactory.CreateOneHotEncoding(option), SweepableEstimatorFactory.CreateOneHotHashEncoding(option) };
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="data"/>. This function will collect all columns in <paramref name="data"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="catalogColumns">columns that should be treated as catalog. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="numericColumns">columns that should be treated as numeric. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="textColumns">columns that should be treated as text. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
internal MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] excludeColumns = null)
public MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] numericColumns = null, string[] textColumns = null, string[] excludeColumns = null)
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));

// validate if there's overlapping among catalogColumns, numericColumns, textColumns and excludeColumns
var overallColumns = new string[][] { catalogColumns, numericColumns, textColumns, excludeColumns }
.Where(c => c != null)
.SelectMany(c => c);

if (overallColumns != null)
{
Contracts.Assert(overallColumns.Count() == overallColumns.Distinct().Count(), "detect overlapping among catalogColumns, numericColumns, textColumns and excludedColumns");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Nit: changed detect overlapping to detected overlapping.

I'm also personally a fan of the oxford comma.

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Resolved

}

var columnInfo = new ColumnInformation();

if (excludeColumns != null)
{
foreach (var ignoreColumn in excludeColumns)
{
columnInfo.IgnoredColumnNames.Add(ignoreColumn);
}
}

if (catalogColumns != null)
{
foreach (var catalogColumn in catalogColumns)
{
columnInfo.CategoricalColumnNames.Add(catalogColumn);
}
}

if (numericColumns != null)
{
foreach (var column in numericColumns)
{
columnInfo.NumericColumnNames.Add(column);
}
}

if (textColumns != null)
{
foreach (var column in textColumns)
{
columnInfo.TextColumnNames.Add(column);
}
}

return this.Featurizer(data, columnInfo, outputColumnName);
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="columnInformation">column information.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
/// <returns></returns>
internal MultiModelPipeline Featurizer(ColumnInformation columnInformation, string outputColumnName = "Features", string[] excludeColumns = null)
/// <returns>A <see cref="MultiModelPipeline"/> for featurization.</returns>
public MultiModelPipeline Featurizer(IDataView data, ColumnInformation columnInformation, string outputColumnName = "Features")
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));
Contracts.CheckValue(columnInformation, nameof(columnInformation));

var columnPurposes = PurposeInference.InferPurposes(this._context, data, columnInformation);
var textFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.TextFeature);
var numericFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.NumericFeature);
var catalogFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.CategoricalFeature);
var textFeatureColumnNames = textFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var numericFeatureColumnNames = numericFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var catalogFeatureColumnNames = catalogFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();

var pipeline = new MultiModelPipeline();
if (numericFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.NumericFeaturizer(numericFeatureColumnNames, numericFeatureColumnNames));
}

if (catalogFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.CatalogFeaturizer(catalogFeatureColumnNames, catalogFeatureColumnNames));
}

foreach (var textColumn in textFeatureColumnNames)
{
pipeline = pipeline.Append(this.TextFeaturizer(textColumn, textColumn));
}

var option = new ConcatOption
{
InputColumnNames = textFeatureColumnNames.Concat(numericFeatureColumnNames).Concat(catalogFeatureColumnNames).ToArray(),
OutputColumnName = outputColumnName,
};

if (option.InputColumnNames.Length > 0)
{
pipeline = pipeline.Append(SweepableEstimatorFactory.CreateConcatenate(option));
}

return pipeline;
}
}
}
2 changes: 1 addition & 1 deletion src/Microsoft.ML.Data/Transforms/Hashing.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -182,7 +182,7 @@ internal HashingTransformer(IHostEnvironment env, params HashingEstimator.Column
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");

if (column.Combine && column.UseOrderedHashing)
throw Host.ExceptParam(nameof(HashingEstimator.ColumnOptions.Combine), "When the 'Combine' option is specified, ordered hashing is not supported.");
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -184,7 +184,7 @@ internal NgramHashingTransformer(IHostEnvironment env, params NgramHashingEstima
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
}
}

Expand Down
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,34 @@
{
"schema": "e0 * e1",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
]
}
},
"e1": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"OutputColumnName": "Features"
}
}
}
}
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,83 @@
{
"schema": "e0 * (e1 \u002B e2) * e3",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"Features"
],
"InputColumnNames": [
"Features"
]
}
},
"e1": {
"estimatorType": "OneHotEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e2": {
"estimatorType": "OneHotHashEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e3": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"Features",
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"OutputColumnName": "OutputFeature"
}
}
}
}
65 changes: 65 additions & 0 deletions test/Microsoft.ML.AutoML.Tests/AutoFeaturizerTests.cs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,65 @@
// Licensed to the .NET Foundation under one or more agreements.
// The .NET Foundation licenses this file to you under the MIT license.
// See the LICENSE file in the project root for more information.

using System;
using System.Collections.Generic;
using System.Text;
using System.Text.Json;
using Microsoft.ML.TestFramework;
using Xunit;
using Xunit.Abstractions;
using ApprovalTests;
using ApprovalTests.Namers;
using ApprovalTests.Reporters;
using System.Text.Json.Serialization;

namespace Microsoft.ML.AutoML.Test
{
public class AutoFeaturizerTests : BaseTestClass
{
private readonly JsonSerializerOptions _jsonSerializerOptions;

public AutoFeaturizerTests(ITestOutputHelper output)
: base(output)
{
_jsonSerializerOptions = new JsonSerializerOptions()
{
WriteIndented = true,
Converters =
{
new JsonStringEnumConverter(), new DoubleToDecimalConverter(), new FloatToDecimalConverter(),
},
};

if (Environment.GetEnvironmentVariable("HELIX_CORRELATION_ID") != null)
{
Approvals.UseAssemblyLocationForApprovedFiles();
}
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_uci_adult_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetUciAdultDataView();
var pipeline = context.Auto().Featurizer(dataset, outputColumnName: "OutputFeature", excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_iris_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetIrisDataView();
var pipeline = context.Auto().Featurizer(dataset, excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}
}
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
148 changes: 130 additions & 18 deletions src/Microsoft.ML.AutoML/API/AutoCatalog.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,8 +4,11 @@

using System;
using System.Collections.Generic;
using System.Diagnostics.Contracts;
using System.Linq;
using Microsoft.ML.AutoML.CodeGen;
using Microsoft.ML.Data;
using Microsoft.ML.Runtime;
using Microsoft.ML.SearchSpace;
using Microsoft.ML.Trainers.FastTree;

Expand DownExpand Up@@ -538,55 +541,164 @@ public SweepableEstimator[] Regression(string labelColumnName = DefaultColumnNam
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] TextFeaturizer(string outputColumnName, string inputColumnName)
{
throw new NotImplementedException();
var option = new FeaturizeTextOption
{
InputColumnName = inputColumnName,
OutputColumnName = outputColumnName,
};

return new[] { SweepableEstimatorFactory.CreateFeaturizeText(option) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing numeric columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] NumericFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] NumericFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.CheckValue(inputColumnNames, nameof(inputColumnNames));
Contracts.CheckValue(outputColumnNames, nameof(outputColumnNames));
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this will blow up if either of those inputs are null. Make sure they aren't null first.

var replaceMissingValueOption = new ReplaceMissingValueOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new[] { SweepableEstimatorFactory.CreateReplaceMissingValues(replaceMissingValueOption) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing catalog columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] CatalogFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] CatalogFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

var option = new OneHotOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new SweepableEstimator[] { SweepableEstimatorFactory.CreateOneHotEncoding(option), SweepableEstimatorFactory.CreateOneHotHashEncoding(option) };
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="data"/>. This function will collect all columns in <paramref name="data"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="catalogColumns">columns that should be treated as catalog. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="numericColumns">columns that should be treated as numeric. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="textColumns">columns that should be treated as text. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
internal MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] excludeColumns = null)
public MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] numericColumns = null, string[] textColumns = null, string[] excludeColumns = null)
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));

// validate if there's overlapping among catalogColumns, numericColumns, textColumns and excludeColumns
var overallColumns = new string[][] { catalogColumns, numericColumns, textColumns, excludeColumns }
.Where(c => c != null)
.SelectMany(c => c);

if (overallColumns != null)
{
Contracts.Assert(overallColumns.Count() == overallColumns.Distinct().Count(), "detect overlapping among catalogColumns, numericColumns, textColumns and excludedColumns");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Nit: changed detect overlapping to detected overlapping.

I'm also personally a fan of the oxford comma.

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Resolved

}

var columnInfo = new ColumnInformation();

if (excludeColumns != null)
{
foreach (var ignoreColumn in excludeColumns)
{
columnInfo.IgnoredColumnNames.Add(ignoreColumn);
}
}

if (catalogColumns != null)
{
foreach (var catalogColumn in catalogColumns)
{
columnInfo.CategoricalColumnNames.Add(catalogColumn);
}
}

if (numericColumns != null)
{
foreach (var column in numericColumns)
{
columnInfo.NumericColumnNames.Add(column);
}
}

if (textColumns != null)
{
foreach (var column in textColumns)
{
columnInfo.TextColumnNames.Add(column);
}
}

return this.Featurizer(data, columnInfo, outputColumnName);
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="columnInformation">column information.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
/// <returns></returns>
internal MultiModelPipeline Featurizer(ColumnInformation columnInformation, string outputColumnName = "Features", string[] excludeColumns = null)
/// <returns>A <see cref="MultiModelPipeline"/> for featurization.</returns>
public MultiModelPipeline Featurizer(IDataView data, ColumnInformation columnInformation, string outputColumnName = "Features")
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));
Contracts.CheckValue(columnInformation, nameof(columnInformation));

var columnPurposes = PurposeInference.InferPurposes(this._context, data, columnInformation);
var textFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.TextFeature);
var numericFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.NumericFeature);
var catalogFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.CategoricalFeature);
var textFeatureColumnNames = textFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var numericFeatureColumnNames = numericFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var catalogFeatureColumnNames = catalogFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();

var pipeline = new MultiModelPipeline();
if (numericFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.NumericFeaturizer(numericFeatureColumnNames, numericFeatureColumnNames));
}

if (catalogFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.CatalogFeaturizer(catalogFeatureColumnNames, catalogFeatureColumnNames));
}

foreach (var textColumn in textFeatureColumnNames)
{
pipeline = pipeline.Append(this.TextFeaturizer(textColumn, textColumn));
}

var option = new ConcatOption
{
InputColumnNames = textFeatureColumnNames.Concat(numericFeatureColumnNames).Concat(catalogFeatureColumnNames).ToArray(),
OutputColumnName = outputColumnName,
};

if (option.InputColumnNames.Length > 0)
{
pipeline = pipeline.Append(SweepableEstimatorFactory.CreateConcatenate(option));
}

return pipeline;
}
}
}
2 changes: 1 addition & 1 deletion src/Microsoft.ML.Data/Transforms/Hashing.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -182,7 +182,7 @@ internal HashingTransformer(IHostEnvironment env, params HashingEstimator.Column
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");

if (column.Combine && column.UseOrderedHashing)
throw Host.ExceptParam(nameof(HashingEstimator.ColumnOptions.Combine), "When the 'Combine' option is specified, ordered hashing is not supported.");
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -184,7 +184,7 @@ internal NgramHashingTransformer(IHostEnvironment env, params NgramHashingEstima
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
}
}

Expand Down
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,34 @@
{
"schema": "e0 * e1",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
]
}
},
"e1": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"OutputColumnName": "Features"
}
}
}
}
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,83 @@
{
"schema": "e0 * (e1 \u002B e2) * e3",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"Features"
],
"InputColumnNames": [
"Features"
]
}
},
"e1": {
"estimatorType": "OneHotEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e2": {
"estimatorType": "OneHotHashEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e3": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"Features",
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"OutputColumnName": "OutputFeature"
}
}
}
}
65 changes: 65 additions & 0 deletions test/Microsoft.ML.AutoML.Tests/AutoFeaturizerTests.cs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,65 @@
// Licensed to the .NET Foundation under one or more agreements.
// The .NET Foundation licenses this file to you under the MIT license.
// See the LICENSE file in the project root for more information.

using System;
using System.Collections.Generic;
using System.Text;
using System.Text.Json;
using Microsoft.ML.TestFramework;
using Xunit;
using Xunit.Abstractions;
using ApprovalTests;
using ApprovalTests.Namers;
using ApprovalTests.Reporters;
using System.Text.Json.Serialization;

namespace Microsoft.ML.AutoML.Test
{
public class AutoFeaturizerTests : BaseTestClass
{
private readonly JsonSerializerOptions _jsonSerializerOptions;

public AutoFeaturizerTests(ITestOutputHelper output)
: base(output)
{
_jsonSerializerOptions = new JsonSerializerOptions()
{
WriteIndented = true,
Converters =
{
new JsonStringEnumConverter(), new DoubleToDecimalConverter(), new FloatToDecimalConverter(),
},
};

if (Environment.GetEnvironmentVariable("HELIX_CORRELATION_ID") != null)
{
Approvals.UseAssemblyLocationForApprovedFiles();
}
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_uci_adult_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetUciAdultDataView();
var pipeline = context.Auto().Featurizer(dataset, outputColumnName: "OutputFeature", excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_iris_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetIrisDataView();
var pipeline = context.Auto().Featurizer(dataset, excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}
}
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
148 changes: 130 additions & 18 deletions src/Microsoft.ML.AutoML/API/AutoCatalog.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,8 +4,11 @@

using System;
using System.Collections.Generic;
using System.Diagnostics.Contracts;
using System.Linq;
using Microsoft.ML.AutoML.CodeGen;
using Microsoft.ML.Data;
using Microsoft.ML.Runtime;
using Microsoft.ML.SearchSpace;
using Microsoft.ML.Trainers.FastTree;

Expand DownExpand Up@@ -538,55 +541,164 @@ public SweepableEstimator[] Regression(string labelColumnName = DefaultColumnNam
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] TextFeaturizer(string outputColumnName, string inputColumnName)
{
throw new NotImplementedException();
var option = new FeaturizeTextOption
{
InputColumnName = inputColumnName,
OutputColumnName = outputColumnName,
};

return new[] { SweepableEstimatorFactory.CreateFeaturizeText(option) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing numeric columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] NumericFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] NumericFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.CheckValue(inputColumnNames, nameof(inputColumnNames));
Contracts.CheckValue(outputColumnNames, nameof(outputColumnNames));
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this will blow up if either of those inputs are null. Make sure they aren't null first.

var replaceMissingValueOption = new ReplaceMissingValueOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new[] { SweepableEstimatorFactory.CreateReplaceMissingValues(replaceMissingValueOption) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing catalog columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] CatalogFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] CatalogFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

var option = new OneHotOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new SweepableEstimator[] { SweepableEstimatorFactory.CreateOneHotEncoding(option), SweepableEstimatorFactory.CreateOneHotHashEncoding(option) };
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="data"/>. This function will collect all columns in <paramref name="data"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="catalogColumns">columns that should be treated as catalog. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="numericColumns">columns that should be treated as numeric. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="textColumns">columns that should be treated as text. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
internal MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] excludeColumns = null)
public MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] numericColumns = null, string[] textColumns = null, string[] excludeColumns = null)
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));

// validate if there's overlapping among catalogColumns, numericColumns, textColumns and excludeColumns
var overallColumns = new string[][] { catalogColumns, numericColumns, textColumns, excludeColumns }
.Where(c => c != null)
.SelectMany(c => c);

if (overallColumns != null)
{
Contracts.Assert(overallColumns.Count() == overallColumns.Distinct().Count(), "detect overlapping among catalogColumns, numericColumns, textColumns and excludedColumns");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Nit: changed detect overlapping to detected overlapping.

I'm also personally a fan of the oxford comma.

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Resolved

}

var columnInfo = new ColumnInformation();

if (excludeColumns != null)
{
foreach (var ignoreColumn in excludeColumns)
{
columnInfo.IgnoredColumnNames.Add(ignoreColumn);
}
}

if (catalogColumns != null)
{
foreach (var catalogColumn in catalogColumns)
{
columnInfo.CategoricalColumnNames.Add(catalogColumn);
}
}

if (numericColumns != null)
{
foreach (var column in numericColumns)
{
columnInfo.NumericColumnNames.Add(column);
}
}

if (textColumns != null)
{
foreach (var column in textColumns)
{
columnInfo.TextColumnNames.Add(column);
}
}

return this.Featurizer(data, columnInfo, outputColumnName);
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="columnInformation">column information.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
/// <returns></returns>
internal MultiModelPipeline Featurizer(ColumnInformation columnInformation, string outputColumnName = "Features", string[] excludeColumns = null)
/// <returns>A <see cref="MultiModelPipeline"/> for featurization.</returns>
public MultiModelPipeline Featurizer(IDataView data, ColumnInformation columnInformation, string outputColumnName = "Features")
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));
Contracts.CheckValue(columnInformation, nameof(columnInformation));

var columnPurposes = PurposeInference.InferPurposes(this._context, data, columnInformation);
var textFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.TextFeature);
var numericFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.NumericFeature);
var catalogFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.CategoricalFeature);
var textFeatureColumnNames = textFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var numericFeatureColumnNames = numericFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var catalogFeatureColumnNames = catalogFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();

var pipeline = new MultiModelPipeline();
if (numericFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.NumericFeaturizer(numericFeatureColumnNames, numericFeatureColumnNames));
}

if (catalogFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.CatalogFeaturizer(catalogFeatureColumnNames, catalogFeatureColumnNames));
}

foreach (var textColumn in textFeatureColumnNames)
{
pipeline = pipeline.Append(this.TextFeaturizer(textColumn, textColumn));
}

var option = new ConcatOption
{
InputColumnNames = textFeatureColumnNames.Concat(numericFeatureColumnNames).Concat(catalogFeatureColumnNames).ToArray(),
OutputColumnName = outputColumnName,
};

if (option.InputColumnNames.Length > 0)
{
pipeline = pipeline.Append(SweepableEstimatorFactory.CreateConcatenate(option));
}

return pipeline;
}
}
}
2 changes: 1 addition & 1 deletion src/Microsoft.ML.Data/Transforms/Hashing.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -182,7 +182,7 @@ internal HashingTransformer(IHostEnvironment env, params HashingEstimator.Column
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");

if (column.Combine && column.UseOrderedHashing)
throw Host.ExceptParam(nameof(HashingEstimator.ColumnOptions.Combine), "When the 'Combine' option is specified, ordered hashing is not supported.");
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -184,7 +184,7 @@ internal NgramHashingTransformer(IHostEnvironment env, params NgramHashingEstima
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
}
}

Expand Down
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,34 @@
{
"schema": "e0 * e1",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
]
}
},
"e1": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"OutputColumnName": "Features"
}
}
}
}
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,83 @@
{
"schema": "e0 * (e1 \u002B e2) * e3",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"Features"
],
"InputColumnNames": [
"Features"
]
}
},
"e1": {
"estimatorType": "OneHotEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e2": {
"estimatorType": "OneHotHashEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e3": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"Features",
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"OutputColumnName": "OutputFeature"
}
}
}
}
65 changes: 65 additions & 0 deletions test/Microsoft.ML.AutoML.Tests/AutoFeaturizerTests.cs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,65 @@
// Licensed to the .NET Foundation under one or more agreements.
// The .NET Foundation licenses this file to you under the MIT license.
// See the LICENSE file in the project root for more information.

using System;
using System.Collections.Generic;
using System.Text;
using System.Text.Json;
using Microsoft.ML.TestFramework;
using Xunit;
using Xunit.Abstractions;
using ApprovalTests;
using ApprovalTests.Namers;
using ApprovalTests.Reporters;
using System.Text.Json.Serialization;

namespace Microsoft.ML.AutoML.Test
{
public class AutoFeaturizerTests : BaseTestClass
{
private readonly JsonSerializerOptions _jsonSerializerOptions;

public AutoFeaturizerTests(ITestOutputHelper output)
: base(output)
{
_jsonSerializerOptions = new JsonSerializerOptions()
{
WriteIndented = true,
Converters =
{
new JsonStringEnumConverter(), new DoubleToDecimalConverter(), new FloatToDecimalConverter(),
},
};

if (Environment.GetEnvironmentVariable("HELIX_CORRELATION_ID") != null)
{
Approvals.UseAssemblyLocationForApprovedFiles();
}
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_uci_adult_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetUciAdultDataView();
var pipeline = context.Auto().Featurizer(dataset, outputColumnName: "OutputFeature", excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_iris_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetIrisDataView();
var pipeline = context.Auto().Featurizer(dataset, excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}
}
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
148 changes: 130 additions & 18 deletions src/Microsoft.ML.AutoML/API/AutoCatalog.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,8 +4,11 @@

using System;
using System.Collections.Generic;
using System.Diagnostics.Contracts;
using System.Linq;
using Microsoft.ML.AutoML.CodeGen;
using Microsoft.ML.Data;
using Microsoft.ML.Runtime;
using Microsoft.ML.SearchSpace;
using Microsoft.ML.Trainers.FastTree;

Expand DownExpand Up@@ -538,55 +541,164 @@ public SweepableEstimator[] Regression(string labelColumnName = DefaultColumnNam
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] TextFeaturizer(string outputColumnName, string inputColumnName)
{
throw new NotImplementedException();
var option = new FeaturizeTextOption
{
InputColumnName = inputColumnName,
OutputColumnName = outputColumnName,
};

return new[] { SweepableEstimatorFactory.CreateFeaturizeText(option) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing numeric columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] NumericFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] NumericFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.CheckValue(inputColumnNames, nameof(inputColumnNames));
Contracts.CheckValue(outputColumnNames, nameof(outputColumnNames));
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this will blow up if either of those inputs are null. Make sure they aren't null first.

var replaceMissingValueOption = new ReplaceMissingValueOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new[] { SweepableEstimatorFactory.CreateReplaceMissingValues(replaceMissingValueOption) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing catalog columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] CatalogFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] CatalogFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

var option = new OneHotOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new SweepableEstimator[] { SweepableEstimatorFactory.CreateOneHotEncoding(option), SweepableEstimatorFactory.CreateOneHotHashEncoding(option) };
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="data"/>. This function will collect all columns in <paramref name="data"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="catalogColumns">columns that should be treated as catalog. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="numericColumns">columns that should be treated as numeric. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="textColumns">columns that should be treated as text. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
internal MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] excludeColumns = null)
public MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] numericColumns = null, string[] textColumns = null, string[] excludeColumns = null)
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));

// validate if there's overlapping among catalogColumns, numericColumns, textColumns and excludeColumns
var overallColumns = new string[][] { catalogColumns, numericColumns, textColumns, excludeColumns }
.Where(c => c != null)
.SelectMany(c => c);

if (overallColumns != null)
{
Contracts.Assert(overallColumns.Count() == overallColumns.Distinct().Count(), "detect overlapping among catalogColumns, numericColumns, textColumns and excludedColumns");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Nit: changed detect overlapping to detected overlapping.

I'm also personally a fan of the oxford comma.

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Resolved

}

var columnInfo = new ColumnInformation();

if (excludeColumns != null)
{
foreach (var ignoreColumn in excludeColumns)
{
columnInfo.IgnoredColumnNames.Add(ignoreColumn);
}
}

if (catalogColumns != null)
{
foreach (var catalogColumn in catalogColumns)
{
columnInfo.CategoricalColumnNames.Add(catalogColumn);
}
}

if (numericColumns != null)
{
foreach (var column in numericColumns)
{
columnInfo.NumericColumnNames.Add(column);
}
}

if (textColumns != null)
{
foreach (var column in textColumns)
{
columnInfo.TextColumnNames.Add(column);
}
}

return this.Featurizer(data, columnInfo, outputColumnName);
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="columnInformation">column information.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
/// <returns></returns>
internal MultiModelPipeline Featurizer(ColumnInformation columnInformation, string outputColumnName = "Features", string[] excludeColumns = null)
/// <returns>A <see cref="MultiModelPipeline"/> for featurization.</returns>
public MultiModelPipeline Featurizer(IDataView data, ColumnInformation columnInformation, string outputColumnName = "Features")
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));
Contracts.CheckValue(columnInformation, nameof(columnInformation));

var columnPurposes = PurposeInference.InferPurposes(this._context, data, columnInformation);
var textFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.TextFeature);
var numericFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.NumericFeature);
var catalogFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.CategoricalFeature);
var textFeatureColumnNames = textFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var numericFeatureColumnNames = numericFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var catalogFeatureColumnNames = catalogFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();

var pipeline = new MultiModelPipeline();
if (numericFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.NumericFeaturizer(numericFeatureColumnNames, numericFeatureColumnNames));
}

if (catalogFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.CatalogFeaturizer(catalogFeatureColumnNames, catalogFeatureColumnNames));
}

foreach (var textColumn in textFeatureColumnNames)
{
pipeline = pipeline.Append(this.TextFeaturizer(textColumn, textColumn));
}

var option = new ConcatOption
{
InputColumnNames = textFeatureColumnNames.Concat(numericFeatureColumnNames).Concat(catalogFeatureColumnNames).ToArray(),
OutputColumnName = outputColumnName,
};

if (option.InputColumnNames.Length > 0)
{
pipeline = pipeline.Append(SweepableEstimatorFactory.CreateConcatenate(option));
}

return pipeline;
}
}
}
2 changes: 1 addition & 1 deletion src/Microsoft.ML.Data/Transforms/Hashing.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -182,7 +182,7 @@ internal HashingTransformer(IHostEnvironment env, params HashingEstimator.Column
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");

if (column.Combine && column.UseOrderedHashing)
throw Host.ExceptParam(nameof(HashingEstimator.ColumnOptions.Combine), "When the 'Combine' option is specified, ordered hashing is not supported.");
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -184,7 +184,7 @@ internal NgramHashingTransformer(IHostEnvironment env, params NgramHashingEstima
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
}
}

Expand Down
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,34 @@
{
"schema": "e0 * e1",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
]
}
},
"e1": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"OutputColumnName": "Features"
}
}
}
}
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,83 @@
{
"schema": "e0 * (e1 \u002B e2) * e3",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"Features"
],
"InputColumnNames": [
"Features"
]
}
},
"e1": {
"estimatorType": "OneHotEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e2": {
"estimatorType": "OneHotHashEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e3": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"Features",
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"OutputColumnName": "OutputFeature"
}
}
}
}
65 changes: 65 additions & 0 deletions test/Microsoft.ML.AutoML.Tests/AutoFeaturizerTests.cs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,65 @@
// Licensed to the .NET Foundation under one or more agreements.
// The .NET Foundation licenses this file to you under the MIT license.
// See the LICENSE file in the project root for more information.

using System;
using System.Collections.Generic;
using System.Text;
using System.Text.Json;
using Microsoft.ML.TestFramework;
using Xunit;
using Xunit.Abstractions;
using ApprovalTests;
using ApprovalTests.Namers;
using ApprovalTests.Reporters;
using System.Text.Json.Serialization;

namespace Microsoft.ML.AutoML.Test
{
public class AutoFeaturizerTests : BaseTestClass
{
private readonly JsonSerializerOptions _jsonSerializerOptions;

public AutoFeaturizerTests(ITestOutputHelper output)
: base(output)
{
_jsonSerializerOptions = new JsonSerializerOptions()
{
WriteIndented = true,
Converters =
{
new JsonStringEnumConverter(), new DoubleToDecimalConverter(), new FloatToDecimalConverter(),
},
};

if (Environment.GetEnvironmentVariable("HELIX_CORRELATION_ID") != null)
{
Approvals.UseAssemblyLocationForApprovedFiles();
}
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_uci_adult_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetUciAdultDataView();
var pipeline = context.Auto().Featurizer(dataset, outputColumnName: "OutputFeature", excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_iris_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetIrisDataView();
var pipeline = context.Auto().Featurizer(dataset, excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}
}
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
148 changes: 130 additions & 18 deletions src/Microsoft.ML.AutoML/API/AutoCatalog.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,8 +4,11 @@

using System;
using System.Collections.Generic;
using System.Diagnostics.Contracts;
using System.Linq;
using Microsoft.ML.AutoML.CodeGen;
using Microsoft.ML.Data;
using Microsoft.ML.Runtime;
using Microsoft.ML.SearchSpace;
using Microsoft.ML.Trainers.FastTree;

Expand DownExpand Up@@ -538,55 +541,164 @@ public SweepableEstimator[] Regression(string labelColumnName = DefaultColumnNam
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] TextFeaturizer(string outputColumnName, string inputColumnName)
{
throw new NotImplementedException();
var option = new FeaturizeTextOption
{
InputColumnName = inputColumnName,
OutputColumnName = outputColumnName,
};

return new[] { SweepableEstimatorFactory.CreateFeaturizeText(option) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing numeric columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] NumericFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] NumericFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.CheckValue(inputColumnNames, nameof(inputColumnNames));
Contracts.CheckValue(outputColumnNames, nameof(outputColumnNames));
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this will blow up if either of those inputs are null. Make sure they aren't null first.

var replaceMissingValueOption = new ReplaceMissingValueOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new[] { SweepableEstimatorFactory.CreateReplaceMissingValues(replaceMissingValueOption) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing catalog columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] CatalogFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] CatalogFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

var option = new OneHotOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new SweepableEstimator[] { SweepableEstimatorFactory.CreateOneHotEncoding(option), SweepableEstimatorFactory.CreateOneHotHashEncoding(option) };
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="data"/>. This function will collect all columns in <paramref name="data"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="catalogColumns">columns that should be treated as catalog. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="numericColumns">columns that should be treated as numeric. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="textColumns">columns that should be treated as text. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
internal MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] excludeColumns = null)
public MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] numericColumns = null, string[] textColumns = null, string[] excludeColumns = null)
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));

// validate if there's overlapping among catalogColumns, numericColumns, textColumns and excludeColumns
var overallColumns = new string[][] { catalogColumns, numericColumns, textColumns, excludeColumns }
.Where(c => c != null)
.SelectMany(c => c);

if (overallColumns != null)
{
Contracts.Assert(overallColumns.Count() == overallColumns.Distinct().Count(), "detect overlapping among catalogColumns, numericColumns, textColumns and excludedColumns");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Nit: changed detect overlapping to detected overlapping.

I'm also personally a fan of the oxford comma.

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Resolved

}

var columnInfo = new ColumnInformation();

if (excludeColumns != null)
{
foreach (var ignoreColumn in excludeColumns)
{
columnInfo.IgnoredColumnNames.Add(ignoreColumn);
}
}

if (catalogColumns != null)
{
foreach (var catalogColumn in catalogColumns)
{
columnInfo.CategoricalColumnNames.Add(catalogColumn);
}
}

if (numericColumns != null)
{
foreach (var column in numericColumns)
{
columnInfo.NumericColumnNames.Add(column);
}
}

if (textColumns != null)
{
foreach (var column in textColumns)
{
columnInfo.TextColumnNames.Add(column);
}
}

return this.Featurizer(data, columnInfo, outputColumnName);
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="columnInformation">column information.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
/// <returns></returns>
internal MultiModelPipeline Featurizer(ColumnInformation columnInformation, string outputColumnName = "Features", string[] excludeColumns = null)
/// <returns>A <see cref="MultiModelPipeline"/> for featurization.</returns>
public MultiModelPipeline Featurizer(IDataView data, ColumnInformation columnInformation, string outputColumnName = "Features")
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));
Contracts.CheckValue(columnInformation, nameof(columnInformation));

var columnPurposes = PurposeInference.InferPurposes(this._context, data, columnInformation);
var textFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.TextFeature);
var numericFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.NumericFeature);
var catalogFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.CategoricalFeature);
var textFeatureColumnNames = textFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var numericFeatureColumnNames = numericFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var catalogFeatureColumnNames = catalogFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();

var pipeline = new MultiModelPipeline();
if (numericFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.NumericFeaturizer(numericFeatureColumnNames, numericFeatureColumnNames));
}

if (catalogFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.CatalogFeaturizer(catalogFeatureColumnNames, catalogFeatureColumnNames));
}

foreach (var textColumn in textFeatureColumnNames)
{
pipeline = pipeline.Append(this.TextFeaturizer(textColumn, textColumn));
}

var option = new ConcatOption
{
InputColumnNames = textFeatureColumnNames.Concat(numericFeatureColumnNames).Concat(catalogFeatureColumnNames).ToArray(),
OutputColumnName = outputColumnName,
};

if (option.InputColumnNames.Length > 0)
{
pipeline = pipeline.Append(SweepableEstimatorFactory.CreateConcatenate(option));
}

return pipeline;
}
}
}
2 changes: 1 addition & 1 deletion src/Microsoft.ML.Data/Transforms/Hashing.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -182,7 +182,7 @@ internal HashingTransformer(IHostEnvironment env, params HashingEstimator.Column
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");

if (column.Combine && column.UseOrderedHashing)
throw Host.ExceptParam(nameof(HashingEstimator.ColumnOptions.Combine), "When the 'Combine' option is specified, ordered hashing is not supported.");
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -184,7 +184,7 @@ internal NgramHashingTransformer(IHostEnvironment env, params NgramHashingEstima
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
}
}

Expand Down
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,34 @@
{
"schema": "e0 * e1",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
]
}
},
"e1": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"OutputColumnName": "Features"
}
}
}
}
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,83 @@
{
"schema": "e0 * (e1 \u002B e2) * e3",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"Features"
],
"InputColumnNames": [
"Features"
]
}
},
"e1": {
"estimatorType": "OneHotEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e2": {
"estimatorType": "OneHotHashEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e3": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"Features",
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"OutputColumnName": "OutputFeature"
}
}
}
}
65 changes: 65 additions & 0 deletions test/Microsoft.ML.AutoML.Tests/AutoFeaturizerTests.cs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,65 @@
// Licensed to the .NET Foundation under one or more agreements.
// The .NET Foundation licenses this file to you under the MIT license.
// See the LICENSE file in the project root for more information.

using System;
using System.Collections.Generic;
using System.Text;
using System.Text.Json;
using Microsoft.ML.TestFramework;
using Xunit;
using Xunit.Abstractions;
using ApprovalTests;
using ApprovalTests.Namers;
using ApprovalTests.Reporters;
using System.Text.Json.Serialization;

namespace Microsoft.ML.AutoML.Test
{
public class AutoFeaturizerTests : BaseTestClass
{
private readonly JsonSerializerOptions _jsonSerializerOptions;

public AutoFeaturizerTests(ITestOutputHelper output)
: base(output)
{
_jsonSerializerOptions = new JsonSerializerOptions()
{
WriteIndented = true,
Converters =
{
new JsonStringEnumConverter(), new DoubleToDecimalConverter(), new FloatToDecimalConverter(),
},
};

if (Environment.GetEnvironmentVariable("HELIX_CORRELATION_ID") != null)
{
Approvals.UseAssemblyLocationForApprovedFiles();
}
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_uci_adult_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetUciAdultDataView();
var pipeline = context.Auto().Featurizer(dataset, outputColumnName: "OutputFeature", excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_iris_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetIrisDataView();
var pipeline = context.Auto().Featurizer(dataset, excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}
}
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
148 changes: 130 additions & 18 deletions src/Microsoft.ML.AutoML/API/AutoCatalog.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,8 +4,11 @@

using System;
using System.Collections.Generic;
using System.Diagnostics.Contracts;
using System.Linq;
using Microsoft.ML.AutoML.CodeGen;
using Microsoft.ML.Data;
using Microsoft.ML.Runtime;
using Microsoft.ML.SearchSpace;
using Microsoft.ML.Trainers.FastTree;

Expand DownExpand Up@@ -538,55 +541,164 @@ public SweepableEstimator[] Regression(string labelColumnName = DefaultColumnNam
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] TextFeaturizer(string outputColumnName, string inputColumnName)
{
throw new NotImplementedException();
var option = new FeaturizeTextOption
{
InputColumnName = inputColumnName,
OutputColumnName = outputColumnName,
};

return new[] { SweepableEstimatorFactory.CreateFeaturizeText(option) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing numeric columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] NumericFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] NumericFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.CheckValue(inputColumnNames, nameof(inputColumnNames));
Contracts.CheckValue(outputColumnNames, nameof(outputColumnNames));
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this will blow up if either of those inputs are null. Make sure they aren't null first.

var replaceMissingValueOption = new ReplaceMissingValueOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new[] { SweepableEstimatorFactory.CreateReplaceMissingValues(replaceMissingValueOption) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing catalog columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] CatalogFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] CatalogFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

var option = new OneHotOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new SweepableEstimator[] { SweepableEstimatorFactory.CreateOneHotEncoding(option), SweepableEstimatorFactory.CreateOneHotHashEncoding(option) };
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="data"/>. This function will collect all columns in <paramref name="data"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="catalogColumns">columns that should be treated as catalog. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="numericColumns">columns that should be treated as numeric. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="textColumns">columns that should be treated as text. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
internal MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] excludeColumns = null)
public MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] numericColumns = null, string[] textColumns = null, string[] excludeColumns = null)
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));

// validate if there's overlapping among catalogColumns, numericColumns, textColumns and excludeColumns
var overallColumns = new string[][] { catalogColumns, numericColumns, textColumns, excludeColumns }
.Where(c => c != null)
.SelectMany(c => c);

if (overallColumns != null)
{
Contracts.Assert(overallColumns.Count() == overallColumns.Distinct().Count(), "detect overlapping among catalogColumns, numericColumns, textColumns and excludedColumns");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Nit: changed detect overlapping to detected overlapping.

I'm also personally a fan of the oxford comma.

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Resolved

}

var columnInfo = new ColumnInformation();

if (excludeColumns != null)
{
foreach (var ignoreColumn in excludeColumns)
{
columnInfo.IgnoredColumnNames.Add(ignoreColumn);
}
}

if (catalogColumns != null)
{
foreach (var catalogColumn in catalogColumns)
{
columnInfo.CategoricalColumnNames.Add(catalogColumn);
}
}

if (numericColumns != null)
{
foreach (var column in numericColumns)
{
columnInfo.NumericColumnNames.Add(column);
}
}

if (textColumns != null)
{
foreach (var column in textColumns)
{
columnInfo.TextColumnNames.Add(column);
}
}

return this.Featurizer(data, columnInfo, outputColumnName);
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="columnInformation">column information.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
/// <returns></returns>
internal MultiModelPipeline Featurizer(ColumnInformation columnInformation, string outputColumnName = "Features", string[] excludeColumns = null)
/// <returns>A <see cref="MultiModelPipeline"/> for featurization.</returns>
public MultiModelPipeline Featurizer(IDataView data, ColumnInformation columnInformation, string outputColumnName = "Features")
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));
Contracts.CheckValue(columnInformation, nameof(columnInformation));

var columnPurposes = PurposeInference.InferPurposes(this._context, data, columnInformation);
var textFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.TextFeature);
var numericFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.NumericFeature);
var catalogFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.CategoricalFeature);
var textFeatureColumnNames = textFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var numericFeatureColumnNames = numericFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var catalogFeatureColumnNames = catalogFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();

var pipeline = new MultiModelPipeline();
if (numericFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.NumericFeaturizer(numericFeatureColumnNames, numericFeatureColumnNames));
}

if (catalogFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.CatalogFeaturizer(catalogFeatureColumnNames, catalogFeatureColumnNames));
}

foreach (var textColumn in textFeatureColumnNames)
{
pipeline = pipeline.Append(this.TextFeaturizer(textColumn, textColumn));
}

var option = new ConcatOption
{
InputColumnNames = textFeatureColumnNames.Concat(numericFeatureColumnNames).Concat(catalogFeatureColumnNames).ToArray(),
OutputColumnName = outputColumnName,
};

if (option.InputColumnNames.Length > 0)
{
pipeline = pipeline.Append(SweepableEstimatorFactory.CreateConcatenate(option));
}

return pipeline;
}
}
}
2 changes: 1 addition & 1 deletion src/Microsoft.ML.Data/Transforms/Hashing.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -182,7 +182,7 @@ internal HashingTransformer(IHostEnvironment env, params HashingEstimator.Column
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");

if (column.Combine && column.UseOrderedHashing)
throw Host.ExceptParam(nameof(HashingEstimator.ColumnOptions.Combine), "When the 'Combine' option is specified, ordered hashing is not supported.");
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -184,7 +184,7 @@ internal NgramHashingTransformer(IHostEnvironment env, params NgramHashingEstima
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
}
}

Expand Down
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,34 @@
{
"schema": "e0 * e1",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
]
}
},
"e1": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"OutputColumnName": "Features"
}
}
}
}
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,83 @@
{
"schema": "e0 * (e1 \u002B e2) * e3",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"Features"
],
"InputColumnNames": [
"Features"
]
}
},
"e1": {
"estimatorType": "OneHotEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e2": {
"estimatorType": "OneHotHashEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e3": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"Features",
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"OutputColumnName": "OutputFeature"
}
}
}
}
65 changes: 65 additions & 0 deletions test/Microsoft.ML.AutoML.Tests/AutoFeaturizerTests.cs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,65 @@
// Licensed to the .NET Foundation under one or more agreements.
// The .NET Foundation licenses this file to you under the MIT license.
// See the LICENSE file in the project root for more information.

using System;
using System.Collections.Generic;
using System.Text;
using System.Text.Json;
using Microsoft.ML.TestFramework;
using Xunit;
using Xunit.Abstractions;
using ApprovalTests;
using ApprovalTests.Namers;
using ApprovalTests.Reporters;
using System.Text.Json.Serialization;

namespace Microsoft.ML.AutoML.Test
{
public class AutoFeaturizerTests : BaseTestClass
{
private readonly JsonSerializerOptions _jsonSerializerOptions;

public AutoFeaturizerTests(ITestOutputHelper output)
: base(output)
{
_jsonSerializerOptions = new JsonSerializerOptions()
{
WriteIndented = true,
Converters =
{
new JsonStringEnumConverter(), new DoubleToDecimalConverter(), new FloatToDecimalConverter(),
},
};

if (Environment.GetEnvironmentVariable("HELIX_CORRELATION_ID") != null)
{
Approvals.UseAssemblyLocationForApprovedFiles();
}
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_uci_adult_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetUciAdultDataView();
var pipeline = context.Auto().Featurizer(dataset, outputColumnName: "OutputFeature", excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_iris_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetIrisDataView();
var pipeline = context.Auto().Featurizer(dataset, excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}
}
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
148 changes: 130 additions & 18 deletions src/Microsoft.ML.AutoML/API/AutoCatalog.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,8 +4,11 @@

using System;
using System.Collections.Generic;
using System.Diagnostics.Contracts;
using System.Linq;
using Microsoft.ML.AutoML.CodeGen;
using Microsoft.ML.Data;
using Microsoft.ML.Runtime;
using Microsoft.ML.SearchSpace;
using Microsoft.ML.Trainers.FastTree;

Expand DownExpand Up@@ -538,55 +541,164 @@ public SweepableEstimator[] Regression(string labelColumnName = DefaultColumnNam
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] TextFeaturizer(string outputColumnName, string inputColumnName)
{
throw new NotImplementedException();
var option = new FeaturizeTextOption
{
InputColumnName = inputColumnName,
OutputColumnName = outputColumnName,
};

return new[] { SweepableEstimatorFactory.CreateFeaturizeText(option) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing numeric columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] NumericFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] NumericFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.CheckValue(inputColumnNames, nameof(inputColumnNames));
Contracts.CheckValue(outputColumnNames, nameof(outputColumnNames));
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this will blow up if either of those inputs are null. Make sure they aren't null first.

var replaceMissingValueOption = new ReplaceMissingValueOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new[] { SweepableEstimatorFactory.CreateReplaceMissingValues(replaceMissingValueOption) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing catalog columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] CatalogFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] CatalogFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

var option = new OneHotOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new SweepableEstimator[] { SweepableEstimatorFactory.CreateOneHotEncoding(option), SweepableEstimatorFactory.CreateOneHotHashEncoding(option) };
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="data"/>. This function will collect all columns in <paramref name="data"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="catalogColumns">columns that should be treated as catalog. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="numericColumns">columns that should be treated as numeric. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="textColumns">columns that should be treated as text. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
internal MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] excludeColumns = null)
public MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] numericColumns = null, string[] textColumns = null, string[] excludeColumns = null)
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));

// validate if there's overlapping among catalogColumns, numericColumns, textColumns and excludeColumns
var overallColumns = new string[][] { catalogColumns, numericColumns, textColumns, excludeColumns }
.Where(c => c != null)
.SelectMany(c => c);

if (overallColumns != null)
{
Contracts.Assert(overallColumns.Count() == overallColumns.Distinct().Count(), "detect overlapping among catalogColumns, numericColumns, textColumns and excludedColumns");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Nit: changed detect overlapping to detected overlapping.

I'm also personally a fan of the oxford comma.

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Resolved

}

var columnInfo = new ColumnInformation();

if (excludeColumns != null)
{
foreach (var ignoreColumn in excludeColumns)
{
columnInfo.IgnoredColumnNames.Add(ignoreColumn);
}
}

if (catalogColumns != null)
{
foreach (var catalogColumn in catalogColumns)
{
columnInfo.CategoricalColumnNames.Add(catalogColumn);
}
}

if (numericColumns != null)
{
foreach (var column in numericColumns)
{
columnInfo.NumericColumnNames.Add(column);
}
}

if (textColumns != null)
{
foreach (var column in textColumns)
{
columnInfo.TextColumnNames.Add(column);
}
}

return this.Featurizer(data, columnInfo, outputColumnName);
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="columnInformation">column information.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
/// <returns></returns>
internal MultiModelPipeline Featurizer(ColumnInformation columnInformation, string outputColumnName = "Features", string[] excludeColumns = null)
/// <returns>A <see cref="MultiModelPipeline"/> for featurization.</returns>
public MultiModelPipeline Featurizer(IDataView data, ColumnInformation columnInformation, string outputColumnName = "Features")
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));
Contracts.CheckValue(columnInformation, nameof(columnInformation));

var columnPurposes = PurposeInference.InferPurposes(this._context, data, columnInformation);
var textFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.TextFeature);
var numericFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.NumericFeature);
var catalogFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.CategoricalFeature);
var textFeatureColumnNames = textFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var numericFeatureColumnNames = numericFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var catalogFeatureColumnNames = catalogFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();

var pipeline = new MultiModelPipeline();
if (numericFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.NumericFeaturizer(numericFeatureColumnNames, numericFeatureColumnNames));
}

if (catalogFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.CatalogFeaturizer(catalogFeatureColumnNames, catalogFeatureColumnNames));
}

foreach (var textColumn in textFeatureColumnNames)
{
pipeline = pipeline.Append(this.TextFeaturizer(textColumn, textColumn));
}

var option = new ConcatOption
{
InputColumnNames = textFeatureColumnNames.Concat(numericFeatureColumnNames).Concat(catalogFeatureColumnNames).ToArray(),
OutputColumnName = outputColumnName,
};

if (option.InputColumnNames.Length > 0)
{
pipeline = pipeline.Append(SweepableEstimatorFactory.CreateConcatenate(option));
}

return pipeline;
}
}
}
2 changes: 1 addition & 1 deletion src/Microsoft.ML.Data/Transforms/Hashing.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -182,7 +182,7 @@ internal HashingTransformer(IHostEnvironment env, params HashingEstimator.Column
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");

if (column.Combine && column.UseOrderedHashing)
throw Host.ExceptParam(nameof(HashingEstimator.ColumnOptions.Combine), "When the 'Combine' option is specified, ordered hashing is not supported.");
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -184,7 +184,7 @@ internal NgramHashingTransformer(IHostEnvironment env, params NgramHashingEstima
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
}
}

Expand Down
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,34 @@
{
"schema": "e0 * e1",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
]
}
},
"e1": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"OutputColumnName": "Features"
}
}
}
}
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,83 @@
{
"schema": "e0 * (e1 \u002B e2) * e3",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"Features"
],
"InputColumnNames": [
"Features"
]
}
},
"e1": {
"estimatorType": "OneHotEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e2": {
"estimatorType": "OneHotHashEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e3": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"Features",
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"OutputColumnName": "OutputFeature"
}
}
}
}
65 changes: 65 additions & 0 deletions test/Microsoft.ML.AutoML.Tests/AutoFeaturizerTests.cs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,65 @@
// Licensed to the .NET Foundation under one or more agreements.
// The .NET Foundation licenses this file to you under the MIT license.
// See the LICENSE file in the project root for more information.

using System;
using System.Collections.Generic;
using System.Text;
using System.Text.Json;
using Microsoft.ML.TestFramework;
using Xunit;
using Xunit.Abstractions;
using ApprovalTests;
using ApprovalTests.Namers;
using ApprovalTests.Reporters;
using System.Text.Json.Serialization;

namespace Microsoft.ML.AutoML.Test
{
public class AutoFeaturizerTests : BaseTestClass
{
private readonly JsonSerializerOptions _jsonSerializerOptions;

public AutoFeaturizerTests(ITestOutputHelper output)
: base(output)
{
_jsonSerializerOptions = new JsonSerializerOptions()
{
WriteIndented = true,
Converters =
{
new JsonStringEnumConverter(), new DoubleToDecimalConverter(), new FloatToDecimalConverter(),
},
};

if (Environment.GetEnvironmentVariable("HELIX_CORRELATION_ID") != null)
{
Approvals.UseAssemblyLocationForApprovedFiles();
}
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_uci_adult_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetUciAdultDataView();
var pipeline = context.Auto().Featurizer(dataset, outputColumnName: "OutputFeature", excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_iris_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetIrisDataView();
var pipeline = context.Auto().Featurizer(dataset, excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}
}
}
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
148 changes: 130 additions & 18 deletions src/Microsoft.ML.AutoML/API/AutoCatalog.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,8 +4,11 @@

using System;
using System.Collections.Generic;
using System.Diagnostics.Contracts;
using System.Linq;
using Microsoft.ML.AutoML.CodeGen;
using Microsoft.ML.Data;
using Microsoft.ML.Runtime;
using Microsoft.ML.SearchSpace;
using Microsoft.ML.Trainers.FastTree;

Expand DownExpand Up@@ -538,55 +541,164 @@ public SweepableEstimator[] Regression(string labelColumnName = DefaultColumnNam
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] TextFeaturizer(string outputColumnName, string inputColumnName)
{
throw new NotImplementedException();
var option = new FeaturizeTextOption
{
InputColumnName = inputColumnName,
OutputColumnName = outputColumnName,
};

return new[] { SweepableEstimatorFactory.CreateFeaturizeText(option) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing numeric columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] NumericFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] NumericFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.CheckValue(inputColumnNames, nameof(inputColumnNames));
Contracts.CheckValue(outputColumnNames, nameof(outputColumnNames));
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this will blow up if either of those inputs are null. Make sure they aren't null first.

var replaceMissingValueOption = new ReplaceMissingValueOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new[] { SweepableEstimatorFactory.CreateReplaceMissingValues(replaceMissingValueOption) };
}

/// <summary>
/// Create a list of <see cref="SweepableEstimator"/> for featurizing catalog columns.
/// </summary>
/// <param name="outputColumnName">output column name.</param>
/// <param name="inputColumnName">input column name.</param>
internal SweepableEstimator[] CatalogFeaturizer(string outputColumnName, string inputColumnName)
/// <param name="outputColumnNames">output column names.</param>
/// <param name="inputColumnNames">input column names.</param>
internal SweepableEstimator[] CatalogFeaturizer(string[] outputColumnNames, string[] inputColumnNames)
{
throw new NotImplementedException();
Contracts.Check(outputColumnNames.Count() == inputColumnNames.Count() && outputColumnNames.Count() > 0, "outputColumnNames and inputColumnNames must have the same length and greater than 0");

var option = new OneHotOption
{
InputColumnNames = inputColumnNames,
OutputColumnNames = outputColumnNames,
};

return new SweepableEstimator[] { SweepableEstimatorFactory.CreateOneHotEncoding(option), SweepableEstimatorFactory.CreateOneHotHashEncoding(option) };
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="data"/>. This function will collect all columns in <paramref name="data"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="catalogColumns">columns that should be treated as catalog. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="numericColumns">columns that should be treated as numeric. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="textColumns">columns that should be treated as text. If not specified, it will automatically infer if a column is catalog or not.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
internal MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] excludeColumns = null)
public MultiModelPipeline Featurizer(IDataView data, string outputColumnName = "Features", string[] catalogColumns = null, string[] numericColumns = null, string[] textColumns = null, string[] excludeColumns = null)
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));

// validate if there's overlapping among catalogColumns, numericColumns, textColumns and excludeColumns
var overallColumns = new string[][] { catalogColumns, numericColumns, textColumns, excludeColumns }
.Where(c => c != null)
.SelectMany(c => c);

if (overallColumns != null)
{
Contracts.Assert(overallColumns.Count() == overallColumns.Distinct().Count(), "detect overlapping among catalogColumns, numericColumns, textColumns and excludedColumns");

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Nit: changed detect overlapping to detected overlapping.

I'm also personally a fan of the oxford comma.

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Resolved

}

var columnInfo = new ColumnInformation();

if (excludeColumns != null)
{
foreach (var ignoreColumn in excludeColumns)
{
columnInfo.IgnoredColumnNames.Add(ignoreColumn);
}
}

if (catalogColumns != null)
{
foreach (var catalogColumn in catalogColumns)
{
columnInfo.CategoricalColumnNames.Add(catalogColumn);
}
}

if (numericColumns != null)
{
foreach (var column in numericColumns)
{
columnInfo.NumericColumnNames.Add(column);
}
}

if (textColumns != null)
{
foreach (var column in textColumns)
{
columnInfo.TextColumnNames.Add(column);
}
}

return this.Featurizer(data, columnInfo, outputColumnName);
}

/// <summary>
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/> and not in <paramref name="excludeColumns"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string, string)"/>, <see cref="NumericFeaturizer(string, string)"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// Create a single featurize pipeline according to <paramref name="columnInformation"/>. This function will collect all columns in <paramref name="columnInformation"/>,
/// featurizing them using <see cref="CatalogFeaturizer(string[], string[])"/>, <see cref="NumericFeaturizer(string[], string[])"/> or <see cref="TextFeaturizer(string, string)"/>. And combine
/// them into a single feature column as output.
/// </summary>
/// <param name="data">input data.</param>
/// <param name="columnInformation">column information.</param>
/// <param name="outputColumnName">output feature column.</param>
/// <param name="excludeColumns">columns that won't be included when featurizing, like label</param>
/// <returns></returns>
internal MultiModelPipeline Featurizer(ColumnInformation columnInformation, string outputColumnName = "Features", string[] excludeColumns = null)
/// <returns>A <see cref="MultiModelPipeline"/> for featurization.</returns>
public MultiModelPipeline Featurizer(IDataView data, ColumnInformation columnInformation, string outputColumnName = "Features")
{
throw new NotImplementedException();
Contracts.CheckValue(data, nameof(data));
Contracts.CheckValue(columnInformation, nameof(columnInformation));

var columnPurposes = PurposeInference.InferPurposes(this._context, data, columnInformation);
var textFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.TextFeature);
var numericFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.NumericFeature);
var catalogFeatures = columnPurposes.Where(c => c.Purpose == ColumnPurpose.CategoricalFeature);
var textFeatureColumnNames = textFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var numericFeatureColumnNames = numericFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();
var catalogFeatureColumnNames = catalogFeatures.Select(c => data.Schema[c.ColumnIndex].Name).ToArray();

var pipeline = new MultiModelPipeline();
if (numericFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.NumericFeaturizer(numericFeatureColumnNames, numericFeatureColumnNames));
}

if (catalogFeatureColumnNames.Length > 0)
{
pipeline = pipeline.Append(this.CatalogFeaturizer(catalogFeatureColumnNames, catalogFeatureColumnNames));
}

foreach (var textColumn in textFeatureColumnNames)
{
pipeline = pipeline.Append(this.TextFeaturizer(textColumn, textColumn));
}

var option = new ConcatOption
{
InputColumnNames = textFeatureColumnNames.Concat(numericFeatureColumnNames).Concat(catalogFeatureColumnNames).ToArray(),
OutputColumnName = outputColumnName,
};

if (option.InputColumnNames.Length > 0)
{
pipeline = pipeline.Append(SweepableEstimatorFactory.CreateConcatenate(option));
}

return pipeline;
}
}
}
2 changes: 1 addition & 1 deletion src/Microsoft.ML.Data/Transforms/Hashing.cs
Original file line numberDiff line numberDiff line change
Expand Up@@ -182,7 +182,7 @@ internal HashingTransformer(IHostEnvironment env, params HashingEstimator.Column
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found column with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(HashingEstimator)} instead");

if (column.Combine && column.UseOrderedHashing)
throw Host.ExceptParam(nameof(HashingEstimator.ColumnOptions.Combine), "When the 'Combine' option is specified, ordered hashing is not supported.");
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -184,7 +184,7 @@ internal NgramHashingTransformer(IHostEnvironment env, params NgramHashingEstima
foreach (var column in _columns)
{
if (column.MaximumNumberOfInverts != 0)
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
throw Host.ExceptParam(nameof(columns), $"Found colunm with {nameof(column.MaximumNumberOfInverts)} set to non zero value, please use {nameof(NgramHashingEstimator)} instead");
}
}

Expand Down
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,34 @@
{
"schema": "e0 * e1",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
]
}
},
"e1": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"col1",
"col2",
"col3",
"col4"
],
"OutputColumnName": "Features"
}
}
}
}
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,83 @@
{
"schema": "e0 * (e1 \u002B e2) * e3",
"estimators": {
"e0": {
"estimatorType": "ReplaceMissingValues",
"parameter": {
"OutputColumnNames": [
"Features"
],
"InputColumnNames": [
"Features"
]
}
},
"e1": {
"estimatorType": "OneHotEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e2": {
"estimatorType": "OneHotHashEncoding",
"parameter": {
"OutputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"InputColumnNames": [
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
]
}
},
"e3": {
"estimatorType": "Concatenate",
"parameter": {
"InputColumnNames": [
"Features",
"Workclass",
"education",
"marital-status",
"occupation",
"relationship",
"ethnicity",
"sex",
"native-country-region"
],
"OutputColumnName": "OutputFeature"
}
}
}
}
65 changes: 65 additions & 0 deletions test/Microsoft.ML.AutoML.Tests/AutoFeaturizerTests.cs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,65 @@
// Licensed to the .NET Foundation under one or more agreements.
// The .NET Foundation licenses this file to you under the MIT license.
// See the LICENSE file in the project root for more information.

using System;
using System.Collections.Generic;
using System.Text;
using System.Text.Json;
using Microsoft.ML.TestFramework;
using Xunit;
using Xunit.Abstractions;
using ApprovalTests;
using ApprovalTests.Namers;
using ApprovalTests.Reporters;
using System.Text.Json.Serialization;

namespace Microsoft.ML.AutoML.Test
{
public class AutoFeaturizerTests : BaseTestClass
{
private readonly JsonSerializerOptions _jsonSerializerOptions;

public AutoFeaturizerTests(ITestOutputHelper output)
: base(output)
{
_jsonSerializerOptions = new JsonSerializerOptions()
{
WriteIndented = true,
Converters =
{
new JsonStringEnumConverter(), new DoubleToDecimalConverter(), new FloatToDecimalConverter(),
},
};

if (Environment.GetEnvironmentVariable("HELIX_CORRELATION_ID") != null)
{
Approvals.UseAssemblyLocationForApprovedFiles();
}
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_uci_adult_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetUciAdultDataView();
var pipeline = context.Auto().Featurizer(dataset, outputColumnName: "OutputFeature", excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}

[Fact]
[UseReporter(typeof(DiffReporter))]
[UseApprovalSubdirectory("ApprovalTests")]
public void AutoFeaturizer_iris_test()
{
var context = new MLContext(1);
var dataset = DatasetUtil.GetIrisDataView();
var pipeline = context.Auto().Featurizer(dataset, excludeColumns: new[] { "Label" });

Approvals.Verify(JsonSerializer.Serialize(pipeline, _jsonSerializerOptions));
}
}
}
Loading