You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

ML.NET构建含可变长度嵌套对象模型的解决方案咨询

ML.NET处理可变长度子对象列表的解决方案

问题描述

我正在使用ML.NET构建机器学习模型,但无法基于包含可变长度子对象列表的复杂对象创建模型。尝试通过数组和VBuffer扁平化数据后均失败:

  • 使用FlattenedModel时抛出错误:Schema mismatch for input column 'SomeFirstFeatures': expected known-size vector or scalar, got VarVector<Single>
  • 使用FlattenedModel2时抛出错误:Schema mismatch for feature column 'Features': expected Vector<Single>, got VarVector<Single>

已找到运行时未知长度的解决方案,但仍缺乏可变长度向量的处理办法,相关代码如下:

public void Test()
{
   List<SomeModel> data = new List<SomeModel>()
   {
        new SomeModel //Note first object has 3 'OtherModel's
        {
            SomeLabel = 1,
            Features = new List<OtherModel>
            {
                new OtherModel
                {
                    SomeFirstFeature = 1,
                    SomeOtherFeature = 2,
                    SomeThirdFeature = 3
                },
                new OtherModel
                {
                    SomeFirstFeature = 4,
                    SomeOtherFeature = 5,
                    SomeThirdFeature = 6
                },
                new OtherModel
                {
                    SomeFirstFeature = 4,
                    SomeOtherFeature = 5,
                    SomeThirdFeature = 6
                }
            }
        },
        new SomeModel //Note second object has 2 'OtherModel's
        {
            SomeLabel = 2,
            Features = new List<OtherModel>
            {
                new OtherModel
                {
                    SomeFirstFeature = 7,
                    SomeOtherFeature = 8,
                    SomeThirdFeature = 9
                },
                new OtherModel
                {
                    SomeFirstFeature = 10,
                    SomeOtherFeature = 11,
                    SomeThirdFeature = 12
                }
            }
        }
    };

    
    //I have tried flattening the data, since ml.net does not support nested data.
    var flattenedData = data.Select(x => new FlattenedModel
    {
        SomeLabel = x.SomeLabel,
        SomeFirstFeatures = x.Features.Select(y => y.SomeFirstFeature).ToArray(),
        SomeOtherFeatures = x.Features.Select(y => y.SomeOtherFeature).ToArray(),
        SomeThirdFeatures = x.Features.Select(y => y.SomeThirdFeature).ToArray()
    }).ToList();

    var flattenedData2 = data.Select(x => new FlattenedModel2
    {
        SomeLabel = x.SomeLabel,
        SomeFirstFeatures = new VBuffer<float>(x.Features.Count ,x.Features.Select(y => y.SomeFirstFeature).ToArray()),
        SomeOtherFeatures = new VBuffer<float>(x.Features.Count, x.Features.Select(y => y.SomeOtherFeature).ToArray()),
        SomeThirdFeatures = new VBuffer<float>(x.Features.Count, x.Features.Select(y => y.SomeThirdFeature).ToArray())
    }).ToList();

    var mlContext = new Microsoft.ML.MLContext();
    IDataView dataView = mlContext.Data.LoadFromEnumerable<FlattenedModel2>(flattenedData2);

    var pipeline = mlContext.Transforms.Concatenate("Features", "SomeFirstFeatures", "SomeOtherFeatures", "SomeThirdFeatures")
         .Append(mlContext.Transforms.NormalizeMinMax("SomeFirstFeatures"))
         .Append(mlContext.Transforms.NormalizeMinMax("SomeOtherFeatures"))
         .Append(mlContext.Transforms.NormalizeMinMax("SomeThirdFeatures"))
         .Append(mlContext.Regression.Trainers.FastTree(labelColumnName: "Label"));
    
    var model = pipeline.Fit(dataView);
    //If i use 'FlattenedModel':
    //Throws
    //System.ArgumentOutOfRangeException:
    //'Schema mismatch for input column 'SomeFirstFeatures':
    //expected known-size vector or scalar, got VarVector<Single>
    //Arg_ParamName_Name'

    //If i use 'FlattenedModel2':
    //System.ArgumentOutOfRangeException:
    //'Schema mismatch for feature column 'Features':
    //expected Vector<Single>, got VarVector<Single> Arg_ParamName_Name'
}

public class SomeModel
{
    [Column("label")]
    public float SomeLabel { get; set; }
    public List<OtherModel> Features { get; set; }
}

public class OtherModel
{
    public float SomeFirstFeature { get; set; }
    public float SomeOtherFeature { get; set; }
    public float SomeThirdFeature { get; set; }
}

public class FlattenedModel
{
    [Column("label")]
    public float SomeLabel { get; set; }
    [VectorType()]
    public float[] SomeFirstFeatures { get; set; }
    [VectorType()]
    public float[] SomeOtherFeatures { get; set; }
    [VectorType()]
    public float[] SomeThirdFeatures { get; set; }
} 

public class FlattenedModel2
{
    [Column("label")]
    public float SomeLabel { get; set; }
    [VectorType()]
    public VBuffer<float> SomeFirstFeatures { get; set; }
    [VectorType()]
    public VBuffer<float> SomeOtherFeatures { get; set; }
    [VectorType()]
    public VBuffer<float> SomeThirdFeatures { get; set; }
}

可行解决方案

ML.NET的传统监督学习算法(如FastTree)仅支持固定长度的向量或标量特征,不兼容可变长度向量(VarVector)。针对你的场景,有两种主流解决思路:

思路1:将可变长度序列转换为固定长度统计特征

对每个可变长度的特征列,计算其统计量(均值、最大值、最小值、总和、标准差等),把这些统计量作为固定长度的输入特征,适配传统算法的要求。

示例代码修改如下:

public class StatFlattenedModel
{
    [Column("label")]
    public float SomeLabel { get; set; }
    
    // SomeFirstFeature的统计特征
    public float SomeFirstFeature_Mean { get; set; }
    public float SomeFirstFeature_Max { get; set; }
    public float SomeFirstFeature_Min { get; set; }
    public float SomeFirstFeature_Sum { get; set; }
    
    // SomeOtherFeature的统计特征
    public float SomeOtherFeature_Mean { get; set; }
    public float SomeOtherFeature_Max { get; set; }
    public float SomeOtherFeature_Min { get; set; }
    public float SomeOtherFeature_Sum { get; set; }
    
    // SomeThirdFeature的统计特征
    public float SomeThirdFeature_Mean { get; set; }
    public float SomeThirdFeature_Max { get; set; }
    public float SomeThirdFeature_Min { get; set; }
    public float SomeThirdFeature_Sum { get; set; }
}

public void TestWithStatFeatures()
{
    List<SomeModel> data = new List<SomeModel>()
    {
        // 原数据保持不变
        new SomeModel
        {
            SomeLabel = 1,
            Features = new List<OtherModel>
            {
                new OtherModel { SomeFirstFeature = 1, SomeOtherFeature = 2, SomeThirdFeature = 3 },
                new OtherModel { SomeFirstFeature = 4, SomeOtherFeature = 5, SomeThirdFeature = 6 },
                new OtherModel { SomeFirstFeature = 4, SomeOtherFeature = 5, SomeThirdFeature = 6 }
            }
        },
        new SomeModel
        {
            SomeLabel = 2,
            Features = new List<OtherModel>
            {
                new OtherModel { SomeFirstFeature = 7, SomeOtherFeature = 8, SomeThirdFeature = 9 },
                new OtherModel { SomeFirstFeature = 10, SomeOtherFeature = 11, SomeThirdFeature = 12 }
            }
        }
    };

    // 转换为统计特征模型
    var statData = data.Select(x => new StatFlattenedModel
    {
        SomeLabel = x.SomeLabel,
        // 计算SomeFirstFeature的统计量
        SomeFirstFeature_Mean = x.Features.Average(y => y.SomeFirstFeature),
        SomeFirstFeature_Max = x.Features.Max(y => y.SomeFirstFeature),
        SomeFirstFeature_Min = x.Features.Min(y => y.SomeFirstFeature),
        SomeFirstFeature_Sum = x.Features.Sum(y => y.SomeFirstFeature),
        // 计算SomeOtherFeature的统计量
        SomeOtherFeature_Mean = x.Features.Average(y => y.SomeOtherFeature),
        SomeOtherFeature_Max = x.Features.Max(y => y.SomeOtherFeature),
        SomeOtherFeature_Min = x.Features.Min(y => y.SomeOtherFeature),
        SomeOtherFeature_Sum = x.Features.Sum(y => y.SomeOtherFeature),
        // 计算SomeThirdFeature的统计量
        SomeThirdFeature_Mean = x.Features.Average(y => y.SomeThirdFeature),
        SomeThirdFeature_Max = x.Features.Max(y => y.SomeThirdFeature),
        SomeThirdFeature_Min = x.Features.Min(y => y.SomeThirdFeature),
        SomeThirdFeature_Sum = x.Features.Sum(y => y.SomeThirdFeature),
    }).ToList();

    var mlContext = new MLContext();
    IDataView dataView = mlContext.Data.LoadFromEnumerable(statData);

    // 拼接所有统计特征为Features列
    var pipeline = mlContext.Transforms.Concatenate("Features", 
        nameof(StatFlattenedModel.SomeFirstFeature_Mean),
        nameof(StatFlattenedModel.SomeFirstFeature_Max),
        nameof(StatFlattenedModel.SomeFirstFeature_Min),
        nameof(StatFlattenedModel.SomeFirstFeature_Sum),
        nameof(StatFlattenedModel.SomeOtherFeature_Mean),
        nameof(StatFlattenedModel.SomeOtherFeature_Max),
        nameof(StatFlattenedModel.SomeOtherFeature_Min),
        nameof(StatFlattenedModel.SomeOtherFeature_Sum),
        nameof(StatFlattenedModel.SomeThirdFeature_Mean),
        nameof(StatFlattenedModel.SomeThirdFeature_Max),
        nameof(StatFlattenedModel.SomeThirdFeature_Min),
        nameof(StatFlattenedModel.SomeThirdFeature_Sum))
        .Append(mlContext.Transforms.NormalizeMinMax("Features"))
        .Append(mlContext.Regression.Trainers.FastTree(labelColumnName: "SomeLabel"));

    var model = pipeline.Fit(dataView);
    // 后续可正常进行预测等操作
}

思路2:使用支持序列输入的模型(如LSTM)

如果序列的顺序信息对预测很重要,可以使用ML.NET中的深度学习组件(如ONNX模型或内置的序列模型),这类模型原生支持可变长度序列输入。不过需要注意:

  1. 需要将序列特征转换为ML.NET支持的序列格式
  2. 回归任务需要适配模型的输出层

示例代码(使用ML.NET的SequenceTransforms):

public void TestWithSequenceModel()
{
    List<SomeModel> data = new List<SomeModel>()
    {
        // 原数据保持不变
        new SomeModel
        {
            SomeLabel = 1,
            Features = new List<OtherModel>
            {
                new OtherModel { SomeFirstFeature = 1, SomeOtherFeature = 2, SomeThirdFeature = 3 },
                new OtherModel { SomeFirstFeature = 4, SomeOtherFeature = 5, SomeThirdFeature = 6 },
                new OtherModel { SomeFirstFeature = 4, SomeOtherFeature = 5, SomeThirdFeature = 6 }
            }
        },
        new SomeModel
        {
            SomeLabel = 2,
            Features = new List<OtherModel>
            {
                new OtherModel { SomeFirstFeature = 7, SomeOtherFeature = 8, SomeThirdFeature = 9 },
                new OtherModel { SomeFirstFeature = 10, SomeOtherFeature = 11, SomeThirdFeature = 12 }
            }
        }
    };

    var mlContext = new MLContext();
    // 直接加载原始数据,无需提前扁平化
    IDataView dataView = mlContext.Data.LoadFromEnumerable(data);

    // 处理嵌套序列:将OtherModel的特征拼接成每个元素的向量,再组成序列
    var pipeline = mlContext.Transforms.Conversion.MapValueToKey("Label", "SomeLabel")
        // 对每个OtherModel实例拼接特征
        .Append(mlContext.Transforms.Concatenate("ElementFeatures", nameof(OtherModel.SomeFirstFeature), nameof(OtherModel.SomeOtherFeature), nameof(OtherModel.SomeThirdFeature))
            .SetInputColumn(nameof(SomeModel.Features)))
        // 将元素向量序列转换为模型可接受的序列格式
        .Append(mlContext.Transforms.NormalizeMinMax("ElementFeatures"))
        // 注意:此处使用多分类示例,回归任务需替换为支持序列的回归模型(如ONNX导入LSTM回归模型)
        .Append(mlContext.MulticlassClassification.Trainers.LbfgsMaximumEntropy(labelColumnName: "Label", featureColumnName: "ElementFeatures"))
        .Append(mlContext.Transforms.Conversion.MapKeyToValue("PredictedLabel"));

    var model = pipeline.Fit(dataView);
}

关键说明

  • 思路1简单易实现,适合序列顺序不重要的场景,完全兼容FastTree等传统算法
  • 思路2适合序列顺序对预测结果有影响的场景,但需要额外配置深度学习相关组件

内容的提问来源于stack exchange,提问作者OneBigQuestion

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.03 11:55:54