You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

能否调整GPT2模型对序列前10个token的注意力分数?

调整GPT2对序列中特定Token的注意力分数方法

当然可以调整Transformer模型对特定Token的注意力分数,attention_mask确实是二进制掩码,只能控制是否关注,无法实现精细的权重调整。以下是几种可行的实现方式:

方法一:通过PyTorch钩子(Hook)修改注意力权重

利用PyTorch的钩子机制,在注意力层计算完原始注意力分数后,对前10个Token的分数进行缩放,从而实现更多/更少关注的效果。

from transformers import AutoTokenizer, GPT2LMHeadModel, AutoConfig
import torch

tokenizer = AutoTokenizer.from_pretrained("gpt2")
device = "cuda" if torch.cuda.is_available() else "cpu"

config = AutoConfig.from_pretrained(
    "gpt2",
    vocab_size=len(tokenizer),
    n_ctx=1024,
    bos_token_id=tokenizer.bos_token_id,
    eos_token_id=tokenizer.eos_token_id,
)

gpt2_model = GPT2LMHeadModel(config).to(device)
my_seq = "Chocolate has a history of human consumption tracing back to 400 AD and is rich in polyphenols such as catechins, anthocyanidins, and pro anthocyanidins. As chocolate and cocoa product consumption, along with interest in them as functional foods, increases worldwide, there is a need to systematically and critically appraise the available clinical evidence on their health effects."
input_ids = tokenizer.encode(my_seq, return_tensors="pt").to(device)

# 定义钩子函数:调整前10个token的注意力分数
def adjust_attention_hook(module, input, output):
    # output[0]是注意力权重(形状:[batch_size, num_heads, seq_len, seq_len])
    attn_weights = output[0]
    # 对前10个token的注意力分数进行缩放(比如乘以2表示更多关注,乘以0.5表示更少关注)
    scale_factor = 2.0  # 可根据需求调整
    # 所有位置对前10个token的注意力分数缩放
    attn_weights[:, :, :, :10] *= scale_factor
    # 前10个位置对所有token的注意力分数也可以缩放(可选)
    # attn_weights[:, :, :10, :] *= scale_factor
    return (attn_weights,) + output[1:]

# 找到GPT2的所有注意力层并注册钩子
for layer in gpt2_model.transformer.h:
    layer.attn.register_forward_hook(adjust_attention_hook)

# 测试生成
output = gpt2_model.generate(
    input_ids=input_ids,
    max_length=150,
    num_return_sequences=1,
    do_sample=False
)
print(tokenizer.decode(output[0], skip_special_tokens=True))

方法二:自定义GPT2注意力层

直接重写GPT2的注意力模块,在计算注意力分数的逻辑中加入自定义的权重调整,这种方式更灵活,适合长期使用。

from transformers import GPT2Attention, GPT2LMHeadModel, AutoConfig, AutoTokenizer
import torch
import torch.nn as nn

# 自定义注意力层
class CustomGPT2Attention(GPT2Attention):
    def __init__(self, config, is_cross_attention=False, layer_idx=None):
        super().__init__(config, is_cross_attention, layer_idx)
        self.adjust_scale = 2.0  # 调整前10个token的缩放系数

    def _attn(self, query, key, value, attention_mask=None, head_mask=None):
        attn_weights = torch.matmul(query, key.transpose(-1, -2))

        if self.scale_attn_weights:
            attn_weights = attn_weights / torch.full(
                [], value.size(-1) ** 0.5, dtype=attn_weights.dtype, device=attn_weights.device
            )

        # 核心:调整前10个token的注意力分数
        if attn_weights.size(-1) >=10:
            attn_weights[:, :, :, :10] *= self.adjust_scale

        # 处理原始注意力掩码逻辑
        if attention_mask is not None:
            attn_weights = attn_weights + attention_mask

        attn_weights = nn.functional.softmax(attn_weights, dim=-1)
        attn_weights = self.attn_dropout(attn_weights)

        if head_mask is not None:
            attn_weights = attn_weights * head_mask

        attn_output = torch.matmul(attn_weights, value)
        return attn_output, attn_weights

# 替换GPT2的注意力层
tokenizer = AutoTokenizer.from_pretrained("gpt2")
device = "cuda" if torch.cuda.is_available() else "cpu"
config = AutoConfig.from_pretrained("gpt2")

gpt2_model = GPT2LMHeadModel(config).to(device)
# 替换每一层的注意力模块
for i in range(config.n_layer):
    gpt2_model.transformer.h[i].attn = CustomGPT2Attention(config, layer_idx=i).to(device)

# 测试
my_seq = "Chocolate has a history of human consumption tracing back to 400 AD and is rich in polyphenols such as catechins, anthocyanidins, and pro anthocyanidins. As chocolate and cocoa product consumption, along with interest in them as functional foods, increases worldwide, there is a need to systematically and critically appraise the available clinical evidence on their health effects."
input_ids = tokenizer.encode(my_seq, return_tensors="pt").to(device)

output = gpt2_model.generate(
    input_ids=input_ids,
    max_length=150,
    num_return_sequences=1,
    do_sample=False
)
print(tokenizer.decode(output[0], skip_special_tokens=True))

方法三:生成阶段手动干预注意力分数(逐Token调整)

如果是在生成过程中需要动态调整,可以在每次生成新Token前,修改模型的注意力权重,适合需要动态调整的场景。

from transformers import AutoTokenizer, GPT2LMHeadModel, AutoConfig
import torch

tokenizer = AutoTokenizer.from_pretrained("gpt2")
device = "cuda" if torch.cuda.is_available() else "cpu"
config = AutoConfig.from_pretrained("gpt2")
gpt2_model = GPT2LMHeadModel(config).to(device)
my_seq = "Chocolate has a history of human consumption tracing back to 400 AD and is rich in polyphenols such as catechins, anthocyanidins, and pro anthocyanidins. As chocolate and cocoa product consumption, along with interest in them as functional foods, increases worldwide, there is a need to systematically and critically appraise the available clinical evidence on their health effects."
input_ids = tokenizer.encode(my_seq, return_tensors="pt").to(device)
current_input = input_ids.clone()

max_length = 150
scale_factor = 2.0

for _ in range(max_length - current_input.size(1)):
    # 前向传播获取注意力权重
    outputs = gpt2_model(current_input, output_attentions=True)
    last_hidden_state = outputs.last_hidden_state
    attentions = outputs.attentions  # 所有层的注意力权重

    # 调整最后一层的注意力权重(也可以调整所有层)
    adjusted_attentions = []
    for attn in attentions:
        if attn.size(-1) >=10:
            attn[:, :, :, :10] *= scale_factor
        adjusted_attentions.append(attn)

    # 基于调整后的注意力计算下一个Token
    logits = outputs.logits
    next_token_id = torch.argmax(logits[:, -1, :], dim=-1).unsqueeze(0)
    current_input = torch.cat([current_input, next_token_id], dim=-1)

print(tokenizer.decode(current_input[0], skip_special_tokens=True))

内容的提问来源于stack exchange,提问作者Penguin

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.07 19:04:55