You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用mmsegmentation训练自定义数据集时遇‘Need at least one array to concatenate’错误

问题

尝试用mmsegmentation训练自定义分割模型,修改配置文件后持续报错:

ValueError: class IterBasedTrainLoop in mmengine/runner/loops.py: class BaseSegDataset in mmseg/datasets/basesegdataset.py: need at least one array to concatenate

排查后发现问题和配置文件的自定义设置(尤其是类名)相关,求解决思路。

完整配置文件

_base_ = [
    '../_base_/models/setr_mla.py', '../_base_/datasets/ade20k.py',
    '../_base_/default_runtime.py', '../_base_/schedules/schedule_160k.py'
]

crop_size = (512, 512)
data_preprocessor = dict(size=crop_size)
norm_cfg = dict(type='SyncBN', requires_grad=True)

num_classes=1

metainfo = dict(classes = ('class_name',),
                palette = [(220, 20, 60),])

dataset_type = 'BaseSegDataset'
data_root = 'path_to_dataset_root_folder'
img_suffix='.png'
seg_map_suffix='.png'

pre_trained_weights_path = 'path_to_weights/weights.pth'

reduce_zero_label = True

model = dict(
    data_preprocessor=data_preprocessor,
    pretrained=None,
    backbone=dict(
        img_size=(512, 512),
        drop_rate=0.,
        init_cfg=dict(
            type='Pretrained', checkpoint=pre_trained_weights_path)),
    decode_head=dict(num_classes=num_classes),
    auxiliary_head=[
        dict(
            type='FCNHead',
            in_channels=256,
            channels=256,
            in_index=0,
            dropout_ratio=0,
            norm_cfg=norm_cfg,
            act_cfg=dict(type='ReLU'),
            num_convs=0,
            kernel_size=1,
            concat_input=False,
            num_classes=num_classes,
            align_corners=False,
            loss_decode=dict(
                type='CrossEntropyLoss', use_sigmoid=False, loss_weight=0.4)),
        dict(
            type='FCNHead',
            in_channels=256,
            channels=256,
            in_index=1,
            dropout_ratio=0,
            norm_cfg=norm_cfg,
            act_cfg=dict(type='ReLU'),
            num_convs=0,
            kernel_size=1,
            concat_input=False,
            num_classes=num_classes,
            align_corners=False,
            loss_decode=dict(
                type='CrossEntropyLoss', use_sigmoid=False, loss_weight=0.4)),
        dict(
            type='FCNHead',
            in_channels=256,
            channels=256,
            in_index=2,
            dropout_ratio=0,
            norm_cfg=norm_cfg,
            act_cfg=dict(type='ReLU'),
            num_convs=0,
            kernel_size=1,
            concat_input=False,
            num_classes=num_classes,
            align_corners=False,
            loss_decode=dict(
                type='CrossEntropyLoss', use_sigmoid=False, loss_weight=0.4)),
        dict(
            type='FCNHead',
            in_channels=256,
            channels=256,
            in_index=3,
            dropout_ratio=0,
            norm_cfg=norm_cfg,
            act_cfg=dict(type='ReLU'),
            num_convs=0,
            kernel_size=1,
            concat_input=False,
            num_classes=num_classes,
            align_corners=False,
            loss_decode=dict(
                type='CrossEntropyLoss', use_sigmoid=False, loss_weight=0.4)),
    ],
    test_cfg=dict(mode='slide', crop_size=(512, 512), stride=(341, 341)),
)

optimizer = dict(lr=0.001, weight_decay=0.0)
optim_wrapper = dict(
    type='OptimWrapper',
    optimizer=optimizer,
    paramwise_cfg=dict(custom_keys={'head': dict(lr_mult=10.)}))

train_dataloader = dict(
    batch_size=8,
    num_workers=8,
    persistent_workers=True,
    sampler=dict(type='InfiniteSampler', shuffle=True),
    dataset=dict(
        type=dataset_type,
        data_root=data_root,
        metainfo=metainfo,
        data_prefix=dict(
            img_path='images/training', 
            seg_map_path='annotations/training'),
        pipeline=[
            dict(type='LoadImageFromFile'),
            dict(type='LoadAnnotations', reduce_zero_label=True),
            dict(
                type='RandomResize',
                scale=(2048, 512),
                ratio_range=(0.5, 2.0),
                keep_ratio=True),
            dict(type='RandomCrop', crop_size=(512, 512), cat_max_ratio=0.75),
            dict(type='RandomFlip', prob=0.5),
            dict(type='PhotoMetricDistortion'),
            dict(type='PackSegInputs')
        ]))

val_dataloader = dict(
    batch_size=8,
    num_workers=8,
    persistent_workers=True,
    sampler=dict(type='DefaultSampler', shuffle=False),
    dataset=dict(
        type=dataset_type,
        data_root=data_root,
        metainfo=metainfo,
        data_prefix=dict(
            img_path='images/validation',
            seg_map_path='annotations/validation'),
        pipeline=[
            dict(type='LoadImageFromFile'),
            dict(type='Resize', scale=(2048, 512), keep_ratio=True),
            dict(type='LoadAnnotations', reduce_zero_label=True),
            dict(type='PackSegInputs')
        ]))

test_dataloader = dict(
    batch_size=8,
    num_workers=8,
    persistent_workers=True,
    sampler=dict(type='DefaultSampler', shuffle=False),
    dataset=dict(
        type=dataset_type,
        data_root=data_root,
        metainfo=metainfo,
        data_prefix=dict(
            img_path='images/test',
            seg_map_path='annotations/test'),
        pipeline=[
            dict(type='LoadImageFromFile'),
            dict(type='Resize', scale=(2048, 512), keep_ratio=True),
            dict(type='LoadAnnotations', reduce_zero_label=True),
            dict(type='PackSegInputs')
        ]))

完整错误日志

Traceback (most recent call last):
  File "/usr/local/lib/python3.8/dist-packages/mmengine/registry/build_functions.py", line 122, in build_from_cfg
    obj = obj_cls(**args)  # type: ignore
  File "/mmsegmentation/mmseg/datasets/basesegdataset.py", line 142, in __init__
    self.full_init()
  File "/usr/local/lib/python3.8/dist-packages/mmengine/dataset/base_dataset.py", line 310, in full_init
    self.data_bytes, self.data_address = self._serialize_data()
  File "/usr/local/lib/python3.8/dist-packages/mmengine/dataset/base_dataset.py", line 772, in _serialize_data
    data_bytes = np.concatenate(data_list)
  File "<__array_function__ internals>", line 180, in concatenate
ValueError: need at least one array to concatenate

During handling of the above exception, another exception occurred:

Traceback (most recent call last):
  File "/usr/local/lib/python3.8/dist-packages/mmengine/registry/build_functions.py", line 122, in build_from_cfg
    obj = obj_cls(**args)  # type: ignore
  File "/usr/local/lib/python3.8/dist-packages/mmengine/runner/loops.py", line 219, in __init__
    super().__init__(runner, dataloader)
  File "/usr/local/lib/python3.8/dist-packages/mmengine/runner/base_loop.py", line 26, in __init__
    self.dataloader = runner.build_dataloader(
  File "/usr/local/lib/python3.8/dist-packages/mmengine/runner/runner.py", line 1346, in build_dataloader
    dataset = DATASETS.build(dataset_cfg)
  File "/usr/local/lib/python3.8/dist-packages/mmengine/registry/registry.py", line 548, in build
    return self.build_func(cfg, *args, **kwargs, registry=self)
  File "/usr/local/lib/python3.8/dist-packages/mmengine/registry/build_functions.py", line 144, in build_from_cfg
    raise type(e)(
ValueError: class `BaseSegDataset` in mmseg/datasets/basesegdataset.py: need at least one array to concatenate

During handling of the above exception, another exception occurred:

Traceback (most recent call last):
  File "/mmsegmentation/tools/train.py", line 104, in <module>
    main()
  File "/mmsegmentation/tools/train.py", line 100, in main
    runner.train()
  File "/usr/local/lib/python3.8/dist-packages/mmengine/runner/runner.py", line 1687, in train
    self._train_loop = self.build_train_loop(
  File "/usr/local/lib/python3.8/dist-packages/mmengine/runner/runner.py", line 1479, in build_train_loop
    loop = LOOPS.build(
  File "/usr/local/lib/python3.8/dist-packages/mmengine/registry/registry.py", line 548, in build
    return self.build_func(cfg, *args, **kwargs, registry=self)
  File "/usr/local/lib/python3.8/dist-packages/mmengine/registry/build_functions.py", line 144, in build_from_cfg
    raise type(e)(
ValueError: class `IterBasedTrainLoop` in mmengine/runner/loops.py: class `BaseSegDataset` in mmseg/datasets/basesegdataset.py: need at least one array to concatenate
srun: error: gr13b04n04: task 0: Exited with exit code 1
解决方案

这个错误本质是数据集加载时未读取到任何有效样本,导致序列化数据时无法拼接空数组,结合配置和场景,从以下几点排查:

  • 检查数据集路径:确认data_root替换为实际绝对路径,data_prefix中的子目录(如images/training)和实际数据集结构一致,目录下确实存在.png后缀的图片和标注文件,无路径拼写错误或文件缺失。
  • 修正单类别分割配置:你设置了num_classes=1且reduce_zero_label=True,这会导致逻辑冲突——reduce_zero_label=True会把标注中的0类当作背景减去,此时有效类别数变为0。单类别场景下应将reduce_zero_label=False,同时确保标注中目标类为1、背景为0;若标注中目标类是0,才需要开启该参数。
  • 更换数据集类:BaseSegDataset是基础类,标准图片-标注配对结构的数据集可尝试将dataset_type改为CustomDataset,确保配置传递了必要参数。
  • 简化数据管道:暂时去掉RandomResize、RandomCrop等增强操作,只保留LoadImageFromFile、LoadAnnotations、PackSegInputs,验证是否能正常加载数据。

内容的提问来源于stack exchange,提问作者Matheus Correia

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.22 05:23:09