Uh oh!
There was an error while loading. Please reload this page.
- Notifications
You must be signed in to change notification settings - Fork 2k
Add FixZero for LogMeanVariance normalizer#3916
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Uh oh!
There was an error while loading. Please reload this page.
Changes from all commits
File filter
Filter by extension
Conversations
Uh oh!
There was an error while loading. Please reload this page.
Jump to
Uh oh!
There was an error while loading. Please reload this page.
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,82 @@ | ||
| using System; | ||
| using System.Collections.Generic; | ||
| using System.Collections.Immutable; | ||
| using System.Linq; | ||
| using Microsoft.ML; | ||
| using Microsoft.ML.Data; | ||
| using static Microsoft.ML.Transforms.NormalizingTransformer; | ||
| namespace Samples.Dynamic | ||
| { | ||
| public class NormalizeLogMeanVarianceFixZero | ||
| { | ||
| public static void Example() | ||
| ||
| { | ||
| // Create a new ML context, for ML.NET operations. It can be used for exception tracking and logging, | ||
| // as well as the source of randomness. | ||
| var mlContext = new MLContext(); | ||
| var samples = new List<DataPoint>() | ||
| { | ||
| new DataPoint(){ Features = new float[5] { 1, 1, 3, 0, float.MaxValue } }, | ||
| new DataPoint(){ Features = new float[5] { 2, 2, 2, 0, float.MinValue } }, | ||
| new DataPoint(){ Features = new float[5] { 0, 0, 1, 0, 0} }, | ||
| new DataPoint(){ Features = new float[5] {-1,-1,-1, 1, 1} } | ||
| }; | ||
| // Convert training data to IDataView, the general data type used in ML.NET. | ||
| var data = mlContext.Data.LoadFromEnumerable(samples); | ||
| // NormalizeLogMeanVariance normalizes the data based on the computed mean and variance of the logarithm of the data. | ||
| // Uses Cumulative distribution function as output. | ||
| var normalize = mlContext.Transforms.NormalizeLogMeanVariance("Features", true, useCdf: true); | ||
| // NormalizeLogMeanVariance normalizes the data based on the computed mean and variance of the logarithm of the data. | ||
| ||
| var normalizeNoCdf = mlContext.Transforms.NormalizeLogMeanVariance("Features", true, useCdf: false); | ||
| // Now we can transform the data and look at the output to confirm the behavior of the estimator. | ||
| // This operation doesn't actually evaluate data until we read the data below. | ||
| var normalizeTransform = normalize.Fit(data); | ||
| var transformedData = normalizeTransform.Transform(data); | ||
| var normalizeNoCdfTransform = normalizeNoCdf.Fit(data); | ||
| var noCdfData = normalizeNoCdfTransform.Transform(data); | ||
| var column = transformedData.GetColumn<float[]>("Features").ToArray(); | ||
| foreach (var row in column) | ||
| Console.WriteLine(string.Join(", ", row.Select(x => x.ToString("f4")))); | ||
| // Expected output: | ||
| // 0.1587, 0.1587, 0.8654, 0.0000, 0.8413 | ||
| // 0.8413, 0.8413, 0.5837, 0.0000, 0.0000 | ||
| // 0.0000, 0.0000, 0.0940, 0.0000, 0.0000 | ||
| // 0.0000, 0.0000, 0.0000, 0.0000, 0.1587 | ||
| var columnFixZero = noCdfData.GetColumn<float[]>("Features").ToArray(); | ||
| foreach (var row in columnFixZero) | ||
| Console.WriteLine(string.Join(", ", row.Select(x => x.ToString("f4")))); | ||
| // Expected output: | ||
| // 2.0403, 2.0403, 4.0001, 0.0000, 5423991000000000000000000000000000000.0000 | ||
| // 4.0806, 4.0806, 2.6667, 0.0000,-5423991000000000000000000000000000000.0000 | ||
| // 0.0000, 0.0000, 1.3334, 0.0000, 0.0000 | ||
| // -2.0403,-2.0403,-1.3334, 0.0000, 0.0159 | ||
| // Let's get transformation parameters. Since we work with only one column we need to pass 0 as parameter for GetNormalizerModelParameters. | ||
| // If we have multiple columns transformations we need to pass index of InputOutputColumnPair. | ||
| var transformParams = normalizeTransform.GetNormalizerModelParameters(0) as CdfNormalizerModelParameters<ImmutableArray<float>>; | ||
| Console.WriteLine("The values in the column with index 1 in the resulting array would be produced by:"); | ||
| Console.WriteLine($"y = 0.5* (1 + ERF((Math.Log(x)- {transformParams.Mean[1]}) / ({transformParams.StandardDeviation[1]} * sqrt(2)))"); | ||
| ||
| // ERF is https://en.wikipedia.org/wiki/Error_function. | ||
| // Expected output: | ||
| // The values in the column with index 1 in the resulting array would be produced by: | ||
| // y = 0.5 * (1 + ERF((Math.Log(x) - 0.3465736) / (0.3465736 * sqrt(2))) | ||
| var noCdfParams = normalizeNoCdfTransform.GetNormalizerModelParameters(0) as AffineNormalizerModelParameters<ImmutableArray<float>>; | ||
| var offset = noCdfParams.Offset.Length == 0 ? 0 : noCdfParams.Offset[1]; | ||
| var scale = noCdfParams.Scale[1]; | ||
| Console.WriteLine($"The values in the column with index 1 in the resulting array would be produced by: y = (x - ({offset})) * {scale}"); | ||
| // Expected output: | ||
| // The values in the column with index 1 in the resulting array would be produced by: y = (x - (0)) * 2.040279 | ||
| } | ||
| private class DataPoint | ||
| { | ||
| [VectorType(5)] | ||
| public float[] Features { get; set; } | ||
| } | ||
| } | ||
| } | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -141,7 +141,7 @@ public static NormalizingEstimator NormalizeLogMeanVariance(this TransformsCatal | ||
| long maximumExampleCount = NormalizingEstimator.Defaults.MaximumExampleCount, | ||
| bool useCdf = NormalizingEstimator.Defaults.LogMeanVarCdf) | ||
| { | ||
| var columnOptions = new NormalizingEstimator.LogMeanVarianceColumnOptions(outputColumnName, inputColumnName, maximumExampleCount, useCdf); | ||
| var columnOptions = new NormalizingEstimator.LogMeanVarianceColumnOptions(outputColumnName, inputColumnName, maximumExampleCount, useCdf, false); | ||
| return new NormalizingEstimator(CatalogUtils.GetEnvironment(catalog), columnOptions); | ||
| } | ||
| @@ -159,7 +159,54 @@ public static NormalizingEstimator NormalizeLogMeanVariance(this TransformsCatal | ||
| bool useCdf = NormalizingEstimator.Defaults.LogMeanVarCdf) => | ||
| new NormalizingEstimator(CatalogUtils.GetEnvironment(catalog), | ||
| columns.Select(column => | ||
| new NormalizingEstimator.LogMeanVarianceColumnOptions(column.OutputColumnName, column.InputColumnName, maximumExampleCount, useCdf)).ToArray()); | ||
| new NormalizingEstimator.LogMeanVarianceColumnOptions(column.OutputColumnName, column.InputColumnName, maximumExampleCount, useCdf, false)).ToArray()); | ||
| /// <summary> | ||
| /// Create a <see cref="NormalizingEstimator"/>, which normalizes based on the computed mean and variance of the logarithm of the data. | ||
| /// </summary> | ||
| /// <param name="catalog">The transform catalog</param> | ||
| /// <param name="outputColumnName">Name of the column resulting from the transformation of <paramref name="inputColumnName"/>. | ||
| /// The data type on this column is the same as the input column.</param> | ||
| /// <param name="fixZero">Whether to map zero to zero, preserving sparsity.</param> | ||
| /// <param name="inputColumnName">Name of the column to transform. If set to <see langword="null"/>, the value of the <paramref name="outputColumnName"/> will be used as source. | ||
| /// The data type on this column should be <see cref="System.Single"/>, <see cref="System.Double"/> or a known-sized vector of those types.</param> | ||
| /// <param name="maximumExampleCount">Maximum number of examples used to train the normalizer.</param> | ||
| /// <param name="useCdf">Whether to use CDF as the output.</param> | ||
| /// <example> | ||
| /// <format type="text/markdown"> | ||
| ||
| /// <] | ||
| /// ]]> | ||
| /// </format> | ||
| /// </example> | ||
| public static NormalizingEstimator NormalizeLogMeanVariance(this TransformsCatalog catalog, | ||
| string outputColumnName, | ||
| bool fixZero, | ||
| string inputColumnName = null, | ||
| long maximumExampleCount = NormalizingEstimator.Defaults.MaximumExampleCount, | ||
| bool useCdf = NormalizingEstimator.Defaults.LogMeanVarCdf) | ||
| { | ||
| var columnOptions = new NormalizingEstimator.LogMeanVarianceColumnOptions(outputColumnName, inputColumnName, maximumExampleCount, useCdf, fixZero); | ||
| return new NormalizingEstimator(CatalogUtils.GetEnvironment(catalog), columnOptions); | ||
| } | ||
| /// <summary> | ||
| /// Create a <see cref="NormalizingEstimator"/>, which normalizes based on the computed mean and variance of the logarithm of the data. | ||
| /// </summary> | ||
| /// <param name="catalog">The transform catalog</param> | ||
| /// <param name="columns">The pairs of input and output columns. | ||
| /// The input columns must be of data type <see cref="System.Single"/>, <see cref="System.Double"/> or a known-sized vector of those types. | ||
| /// The data type for the output column will be the same as the associated input column.</param> | ||
| /// <param name="fixZero">Whether to map zero to zero, preserving sparsity.</param> | ||
| /// <param name="maximumExampleCount">Maximum number of examples used to train the normalizer.</param> | ||
| /// <param name="useCdf">Whether to use CDF as the output.</param> | ||
| public static NormalizingEstimator NormalizeLogMeanVariance(this TransformsCatalog catalog, InputOutputColumnPair[] columns, | ||
| bool fixZero, | ||
| long maximumExampleCount = NormalizingEstimator.Defaults.MaximumExampleCount, | ||
| bool useCdf = NormalizingEstimator.Defaults.LogMeanVarCdf) => | ||
| new NormalizingEstimator(CatalogUtils.GetEnvironment(catalog), | ||
| columns.Select(column => | ||
| new NormalizingEstimator.LogMeanVarianceColumnOptions(column.OutputColumnName, column.InputColumnName, maximumExampleCount, useCdf, fixZero)).ToArray()); | ||
| /// <summary> | ||
| /// Create a <see cref="NormalizingEstimator"/>, which normalizes by assigning the data into bins with equal density. | ||
Uh oh!
There was an error while loading. Please reload this page.
Uh oh!
There was an error while loading. Please reload this page.
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Is this change related to
FixZero? #ResolvedThere was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
It's related to the original issue posted by Lisa. They were encountering an issue with the MeanVariance normalizer which does not handle data that looks like
float.MaxValue,float.MinValue. I think it would be useful to show that LogMeanVariance normalizer accepts this data in the related sample.In reply to: 298254220 [](ancestors = 298254220)