You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Hugging Face ViT模型特征提取:pooler_output与last_hidden_state选哪个?

使用Hugging Face ViT模型提取特征:outputs.pooler_output vs outputs.last_hidden_state[:, 0]

TLDR: 使用Hugging Face(HF)ViT模型提取特征时,正确方式是使用outputs.pooler_output还是outputs.last_hidden_state[:, 0]?其中outputs由outputs: BaseModelOutputWithPooling = self.model(pixel_values=batch_xs)得到。


仅依靠ViT模型解决图像分类问题时,正确的实现方式并不明确。笔者最终采用了如下方案,但不确定是否正确或最优(文末附完整代码):

outputs: BaseModelOutputWithPooling = self.model(pixel_values=batch_xs)
output: Tensor = self.dropout(outputs.last_hidden_state[:, 0])
logits: Tensor = self.cls(output)

从直觉上看,提取cls token位置的特征是合理的。但查看ViTModel的所有层后,笔者倾向于选择(pooler): ViTPooler(...)层中经过Tanh()后的输出,因为它紧邻分类层,且激活值范围看起来更合理。以下是两种方式的输出结果:

outputs: BaseModelOutputWithPooling = self.model(pixel_values=batch_xs)

outputs.pooler_output
tensor([[-0.3976, -0.8454, -0.0601,  ..., -0.2804, -0.1822,  0.1917],
        [-0.3392, -0.0248,  0.1346,  ..., -0.5822,  0.8779,  0.4147],
        [-0.2980, -0.8038, -0.1146,  ...,  0.2431, -0.0963,  0.7844],
        ...,
        [-0.1237, -0.7514,  0.7388,  ..., -0.8551,  0.1512,  0.6157],
        [ 0.5351, -0.9040,  0.0387,  ..., -0.0773,  0.2704, -0.0311],
        [ 0.2142, -0.3138,  0.0426,  ..., -0.5943,  0.2873,  0.4420]],
       grad_fn=<TanhBackward>)
outputs.last_hidden_state[:, 0]
tensor([[ 5.7313e-01, -2.1335e+00,  2.0491e-01,  ..., -1.2373e-01,
         -2.0056e-01, -4.8167e-01],
        [ 5.3309e-02, -1.6563e+00,  1.5719e+00,  ..., -1.3617e+00,
         -3.0064e-01, -2.0056e-01],
        [-2.0633e-02, -2.1370e+00,  9.9927e-01,  ..., -2.3584e+00,
          8.6123e-01, -1.2759e+00],
        ...,
        [ 3.9583e-01, -1.3500e+00,  1.7638e+00,  ..., -9.9536e-01,
          1.0843e+00, -4.4368e-01],
        [ 1.6026e+00, -6.4654e-01,  2.4882e+00,  ..., -1.0347e+00,
         -1.3160e-03, -2.4357e+00],
        [-1.2769e-02, -9.6574e-01,  1.6432e+00,  ..., -7.9090e-01,
          6.1669e-01,  3.2990e-01]], grad_fn=<SelectBackward>)

以下是sanity check的求和结果:

outputs.pooler_output.sum()
tensor(3.8430, grad_fn=<SumBackward0>)
outputs.last_hidden_state[:, 0].sum()
tensor(-6.4373e-06, grad_fn=<SumBackward0>)

形状信息:

outputs.pooler_output.shape
torch.Size([25, 768])
outputs.last_hidden_state[:, 0].shape
torch.Size([25, 768])

outputs.pooler_output的表现看起来更合理,但笔者的前向传播却使用了outputs.last_hidden_state[:, 0]。请问应该选择哪种方式?

完整代码

class ViTForImageClassificationUU(nn.Module):
    def __init__(self,
                 num_classes: int,
                 image_size: int,  # 224 inet, 32 cifar, 84 mi, 28 mnist, omni...
                 criterion: Optional[Union[None, Callable]] = None,
                 # Note: USL agent does criterion not model usually for me e.g nn.Criterion()
                 cls_p_dropout: float = 0.0,
                 pretrained_name: str = None,
                 vitconfig: ViTConfig = None,
                 ):
        """
        :param num_classes:
        :param pretrained_name: 'google/vit-base-patch16-224-in21k'  # 与"google/vit-base-patch16-224"的区别是什么
        """
        super().__init__()
        if vitconfig is not None:
            raise NotImplementedError
            self.vitconfig = vitconfig
            print(f'你传入了配置参数,其他所有参数都会被忽略。')
        elif pretrained_name is not None:
            raise NotImplementedError
            # self.vit = ViTModel.from_pretrained('google/vit-base-patch16-224-in21k')
            self.model = ViTModel.from_pretrained(pretrained_name)
            print('确保你没有传入vitconfig,否则预训练模型名称会被忽略。')
        else:
            self.num_classes = num_classes
            self.image_size = image_size
            self.vitconfig = ViTConfig(image_size=self.image_size)
            self.model = ViTModel(self.vitconfig)
        assert cls_p_dropout == 0.0, '错误:目前分类器仅支持dropout概率为0,待确认是否需要修改其他层的dropout参数。'
        self.dropout = nn.Dropout(cls_p_dropout)
        self.cls = nn.Linear(self.model.config.hidden_size, num_classes)
        self.criterion = None if criterion is None else criterion

    def forward(self, batch_xs: Tensor, labels: Tensor = None) -> Tensor:
        """
        ViT的前向传播。我添加了“缺失”的分类器(以及之前的dropout层)作用于第一个cls token的嵌入,其余token嵌入被忽略。

        我认为特征提取器仅负责数据归一化,似乎不会将数据转换为序列,不过相关教程中仍有使用它的例子。
        """
        outputs: BaseModelOutputWithPooling = self.model(pixel_values=batch_xs)
        output: Tensor = self.dropout(outputs.last_hidden_state[:, 0])
        logits: Tensor = self.cls(output)
        if labels is None:
            assert logits.dtype == torch.float32
            return logits  # 这是我的USL代理的处理方式 ;)
        else:
            raise NotImplementedError
            assert labels.dtype == torch.long
            #   loss = self.criterion(logits.view(-1, self.num_classes), labels.view(-1))
            loss = self.criterion(logits, labels)
            return loss, logits

    def get_embedding(self, batch_xs: Tensor) -> Tensor:
        """
        获取第一个cls token的特征嵌入。

        细节:
        观察ViTLayer可知,(pooler) ViTPooler(...)层包含激活函数和Tanh()层。
        从实验结果看,它的输出更合理:
            outputs.pooler_output.sum()
            tensor(3.8430, grad_fn=<SumBackward0>)
        而手动提取cls位置特征的结果看起来很奇怪:
            outputs.last_hidden_state[:, 0, :].sum()
            tensor(-6.4373e-06, grad_fn=<SumBackward0>)
        """
        outputs: BaseModelOutputWithPooling = self.model(pixel_values=batch_xs)
        feat = outputs.pooler_output
        return feat

    def _assert_its_random_model(self):
        from uutils.torch_uu import norm
        pre_trained_model = ViTModel.from_pretrained('google/vit-base-patch16-224-in21k')
        print(f'----> {norm(pre_trained_model)=}')
        print(f'----> {norm(self)=}')
        assert norm(pre_trained_model) > norm(self), f'随机模型的权重通常更小,但得到的结果是{norm(pre_trained_model)}{norm(self)}'


def get_vit_get_vit_model_and_model_hps(vitconfig: ViTConfig = None,
                                        num_classes: int = 5,
                                        image_size: int = 84,  # 224 inet, 32 cifar, 84 mi, 28 mnist, omni...
                                        criterion: Optional[Union[None, Callable]] = None,  # 我的代理通常负责处理损失计算
                                        cls_p_dropout: float = 0.0,
                                        pretrained_name: str = None,
                                        ) -> tuple[nn.Module, dict]:
    """获取适用于MI任务的ViT模型,仅需指定num_classes=5和image_size=84。"""
    model_hps: dict = dict(vitconfig=vitconfig,
                           num_classes=num_classes,
                           image_size=image_size,
                           criterion=criterion,
                           cls_p_dropout=cls_p_dropout,
                           pretrained_name=pretrained_name)
    model: nn.Module = ViTForImageClassificationUU(**model_hps)
    print('建议为ViT模型设置args.allow_unused = True。')
    return model, model_hps

def vit_forward_pass():
    # 确保结果可复现
    import random
    import numpy as np
    random.seed(0)
    torch.manual_seed(0)
    np.random.seed(0)

    # 设置设备
    device = torch.device(f"cuda:{0}" if torch.cuda.is_available() else "cpu")

    # 获取ViT模型
    vitconfig: ViTConfig = ViTConfig()
    model = get_vit_get_vit_model_and_model_hps(vitconfig, num_classes=64 + 1100, image_size=84)
    criterion = nn.CrossEntropyLoss()
    # 移动到设备
    model.to(device)
    criterion.to(device)

    # 前向传播测试
    x = torch.rand(5, 3, 84, 84)
    y = torch.randint(0, 64 + 1100, (5,))
    logits = model(x)
    loss = criterion(logits, y)
    print(f'{loss=}')

内容的提问来源于stack exchange,提问作者Charlie Parker

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.27 08:44:53