Hugging Face ViT模型特征提取:pooler_output与last_hidden_state选哪个?
TLDR: 使用Hugging Face(HF)ViT模型提取特征时,正确方式是使用outputs.pooler_output还是outputs.last_hidden_state[:, 0]?其中outputs由outputs: BaseModelOutputWithPooling = self.model(pixel_values=batch_xs)得到。
仅依靠ViT模型解决图像分类问题时,正确的实现方式并不明确。笔者最终采用了如下方案,但不确定是否正确或最优(文末附完整代码):
outputs: BaseModelOutputWithPooling = self.model(pixel_values=batch_xs) output: Tensor = self.dropout(outputs.last_hidden_state[:, 0]) logits: Tensor = self.cls(output)
从直觉上看,提取cls token位置的特征是合理的。但查看ViTModel的所有层后,笔者倾向于选择(pooler): ViTPooler(...)层中经过Tanh()后的输出,因为它紧邻分类层,且激活值范围看起来更合理。以下是两种方式的输出结果:
outputs: BaseModelOutputWithPooling = self.model(pixel_values=batch_xs) outputs.pooler_output tensor([[-0.3976, -0.8454, -0.0601, ..., -0.2804, -0.1822, 0.1917], [-0.3392, -0.0248, 0.1346, ..., -0.5822, 0.8779, 0.4147], [-0.2980, -0.8038, -0.1146, ..., 0.2431, -0.0963, 0.7844], ..., [-0.1237, -0.7514, 0.7388, ..., -0.8551, 0.1512, 0.6157], [ 0.5351, -0.9040, 0.0387, ..., -0.0773, 0.2704, -0.0311], [ 0.2142, -0.3138, 0.0426, ..., -0.5943, 0.2873, 0.4420]], grad_fn=<TanhBackward>) outputs.last_hidden_state[:, 0] tensor([[ 5.7313e-01, -2.1335e+00, 2.0491e-01, ..., -1.2373e-01, -2.0056e-01, -4.8167e-01], [ 5.3309e-02, -1.6563e+00, 1.5719e+00, ..., -1.3617e+00, -3.0064e-01, -2.0056e-01], [-2.0633e-02, -2.1370e+00, 9.9927e-01, ..., -2.3584e+00, 8.6123e-01, -1.2759e+00], ..., [ 3.9583e-01, -1.3500e+00, 1.7638e+00, ..., -9.9536e-01, 1.0843e+00, -4.4368e-01], [ 1.6026e+00, -6.4654e-01, 2.4882e+00, ..., -1.0347e+00, -1.3160e-03, -2.4357e+00], [-1.2769e-02, -9.6574e-01, 1.6432e+00, ..., -7.9090e-01, 6.1669e-01, 3.2990e-01]], grad_fn=<SelectBackward>)
以下是sanity check的求和结果:
outputs.pooler_output.sum() tensor(3.8430, grad_fn=<SumBackward0>) outputs.last_hidden_state[:, 0].sum() tensor(-6.4373e-06, grad_fn=<SumBackward0>)
形状信息:
outputs.pooler_output.shape torch.Size([25, 768]) outputs.last_hidden_state[:, 0].shape torch.Size([25, 768])
outputs.pooler_output的表现看起来更合理,但笔者的前向传播却使用了outputs.last_hidden_state[:, 0]。请问应该选择哪种方式?
完整代码
class ViTForImageClassificationUU(nn.Module): def __init__(self, num_classes: int, image_size: int, # 224 inet, 32 cifar, 84 mi, 28 mnist, omni... criterion: Optional[Union[None, Callable]] = None, # Note: USL agent does criterion not model usually for me e.g nn.Criterion() cls_p_dropout: float = 0.0, pretrained_name: str = None, vitconfig: ViTConfig = None, ): """ :param num_classes: :param pretrained_name: 'google/vit-base-patch16-224-in21k' # 与"google/vit-base-patch16-224"的区别是什么 """ super().__init__() if vitconfig is not None: raise NotImplementedError self.vitconfig = vitconfig print(f'你传入了配置参数,其他所有参数都会被忽略。') elif pretrained_name is not None: raise NotImplementedError # self.vit = ViTModel.from_pretrained('google/vit-base-patch16-224-in21k') self.model = ViTModel.from_pretrained(pretrained_name) print('确保你没有传入vitconfig,否则预训练模型名称会被忽略。') else: self.num_classes = num_classes self.image_size = image_size self.vitconfig = ViTConfig(image_size=self.image_size) self.model = ViTModel(self.vitconfig) assert cls_p_dropout == 0.0, '错误:目前分类器仅支持dropout概率为0,待确认是否需要修改其他层的dropout参数。' self.dropout = nn.Dropout(cls_p_dropout) self.cls = nn.Linear(self.model.config.hidden_size, num_classes) self.criterion = None if criterion is None else criterion def forward(self, batch_xs: Tensor, labels: Tensor = None) -> Tensor: """ ViT的前向传播。我添加了“缺失”的分类器(以及之前的dropout层)作用于第一个cls token的嵌入,其余token嵌入被忽略。 我认为特征提取器仅负责数据归一化,似乎不会将数据转换为序列,不过相关教程中仍有使用它的例子。 """ outputs: BaseModelOutputWithPooling = self.model(pixel_values=batch_xs) output: Tensor = self.dropout(outputs.last_hidden_state[:, 0]) logits: Tensor = self.cls(output) if labels is None: assert logits.dtype == torch.float32 return logits # 这是我的USL代理的处理方式 ;) else: raise NotImplementedError assert labels.dtype == torch.long # loss = self.criterion(logits.view(-1, self.num_classes), labels.view(-1)) loss = self.criterion(logits, labels) return loss, logits def get_embedding(self, batch_xs: Tensor) -> Tensor: """ 获取第一个cls token的特征嵌入。 细节: 观察ViTLayer可知,(pooler) ViTPooler(...)层包含激活函数和Tanh()层。 从实验结果看,它的输出更合理: outputs.pooler_output.sum() tensor(3.8430, grad_fn=<SumBackward0>) 而手动提取cls位置特征的结果看起来很奇怪: outputs.last_hidden_state[:, 0, :].sum() tensor(-6.4373e-06, grad_fn=<SumBackward0>) """ outputs: BaseModelOutputWithPooling = self.model(pixel_values=batch_xs) feat = outputs.pooler_output return feat def _assert_its_random_model(self): from uutils.torch_uu import norm pre_trained_model = ViTModel.from_pretrained('google/vit-base-patch16-224-in21k') print(f'----> {norm(pre_trained_model)=}') print(f'----> {norm(self)=}') assert norm(pre_trained_model) > norm(self), f'随机模型的权重通常更小,但得到的结果是{norm(pre_trained_model)}{norm(self)}' def get_vit_get_vit_model_and_model_hps(vitconfig: ViTConfig = None, num_classes: int = 5, image_size: int = 84, # 224 inet, 32 cifar, 84 mi, 28 mnist, omni... criterion: Optional[Union[None, Callable]] = None, # 我的代理通常负责处理损失计算 cls_p_dropout: float = 0.0, pretrained_name: str = None, ) -> tuple[nn.Module, dict]: """获取适用于MI任务的ViT模型,仅需指定num_classes=5和image_size=84。""" model_hps: dict = dict(vitconfig=vitconfig, num_classes=num_classes, image_size=image_size, criterion=criterion, cls_p_dropout=cls_p_dropout, pretrained_name=pretrained_name) model: nn.Module = ViTForImageClassificationUU(**model_hps) print('建议为ViT模型设置args.allow_unused = True。') return model, model_hps def vit_forward_pass(): # 确保结果可复现 import random import numpy as np random.seed(0) torch.manual_seed(0) np.random.seed(0) # 设置设备 device = torch.device(f"cuda:{0}" if torch.cuda.is_available() else "cpu") # 获取ViT模型 vitconfig: ViTConfig = ViTConfig() model = get_vit_get_vit_model_and_model_hps(vitconfig, num_classes=64 + 1100, image_size=84) criterion = nn.CrossEntropyLoss() # 移动到设备 model.to(device) criterion.to(device) # 前向传播测试 x = torch.rand(5, 3, 84, 84) y = torch.randint(0, 64 + 1100, (5,)) logits = model(x) loss = criterion(logits, y) print(f'{loss=}')
内容的提问来源于stack exchange,提问作者Charlie Parker
相关产品推荐
相关产品推荐

