You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

解决GPT-2+LoRA训练IMDb情感分析时的ValueError问题

GPT-2 + LoRA 做IMDb情感分析时出现input_ids缺失错误

我正在基于IMDb数据集和GPT-2模型开发情感分析玩具项目,目标是学习PEFT、LoRA技术并熟悉Huggingface库。以下是我的代码:

from datasets import load_dataset

splits = ["train", "test"]
ds = {split: ds for split, ds in zip(splits, load_dataset("imdb", split=splits))}

from transformers import AutoTokenizer

tokenizer = AutoTokenizer.from_pretrained("gpt2")
# GPT-2 Tokenizer doesn't have a padding token.
tokenizer.pad_token = tokenizer.eos_token

def preprocess_function(examples):
    """Preprocess the imdb dataset by returning tokenized examples."""
    tokens = tokenizer(examples['text'],padding='max_length',truncation=True)
    return tokens


tokenized_ds = {}
for split in splits:
    tokenized_ds[split] = ds[split].map(preprocess_function, batched=True)


model2 = AutoModelForSequenceClassification.from_pretrained(
    "gpt2",
    num_labels=2,
    id2label={0: "NEGATIVE", 1: "POSITIVE"},  # For converting predictions to strings
    label2id={"NEGATIVE": 0, "POSITIVE":1},
)
model2.config.pad_token_id = model.config.eos_token_id

from peft import LoraConfig
from peft import get_peft_model
lora_config = LoraConfig("lora_gpt2", fan_in_fan_out=True,)
lora_model = get_peft_model(model2, lora_config)

trainer_lora = Trainer(
    model=lora_model,
    args=TrainingArguments(
        output_dir="./data/sentiment_analysis2",
        learning_rate=2e-3,
        # Reduce the batch size if you don't have enough memory
        per_device_train_batch_size=4,
        per_device_eval_batch_size=4,
        num_train_epochs=5,
        weight_decay=0.01,
        evaluation_strategy="epoch",
        save_strategy="epoch",
        load_best_model_at_end=True,
    ),
    train_dataset=tokenized_ds["train"],
    eval_dataset=tokenized_ds["test"],
    tokenizer=tokenizer,
    data_collator=DataCollatorWithPadding(tokenizer=tokenizer),
    compute_metrics=compute_metrics,
)

trainer_lora.train()

运行后遇到以下错误:

File /opt/conda/lib/python3.10/site-packages/transformers/tokenization_utils_base.py:3018, in PreTrainedTokenizerBase.pad(self, encoded_inputs, padding, max_length, pad_to_multiple_of, return_attention_mask, return_tensors, verbose)
   3016 # The model's main input name, usually `input_ids`, has be passed for padding
   3017 if self.model_input_names[0] not in encoded_inputs:
-> 3018     raise ValueError(
   3019         "You should supply an encoding or a list of encodings to this method "
   3020         f"that includes {self.model_input_names[0]}, but you provided {list(encoded_inputs.keys())}"
   3021     )
   3023 required_input = encoded_inputs[self.model_input_names[0]]
   3025 if required_input is None or (isinstance(required_input, Sized) and len(required_input) == 0):

ValueError: You should supply an encoding or a list of encodings to this method that includes input_ids, but you provided ['label']

问题原因与修复方案

1. 预处理函数未保留label字段

预处理函数仅返回tokenizer生成的input_ids和attention_mask,丢失了原始数据集的label字段,导致数据加载时字段混乱,最终触发pad方法的错误。

修复: 在预处理函数中保留label字段:

def preprocess_function(examples):
    """Preprocess the imdb dataset by returning tokenized examples."""
    tokens = tokenizer(examples['text'], padding='max_length', truncation=True)
    # 保留原始标签字段
    tokens["label"] = examples["label"]
    return tokens

2. pad_token_id配置错误

代码中model2.config.pad_token_id = model.config.eos_token_id引用了未定义的model变量,应改为使用已定义的model2或直接用tokenizer的pad_token_id:

model2.config.pad_token_id = tokenizer.pad_token_id

3. 移除冗余的DataCollatorWithPadding

预处理阶段已经使用padding='max_length'完成了padding,无需再用DataCollatorWithPadding重复处理,这会导致数据加载逻辑冲突,直接移除该参数即可。

4. 完善LoRA配置

原LoRAConfig缺少任务类型和关键参数,需补充序列分类任务的必要配置,确保LoRA层正确应用到GPT-2的注意力模块:

lora_config = LoraConfig(
    r=8,
    lora_alpha=32,
    target_modules=["c_attn"],  # GPT-2的注意力层
    lora_dropout=0.05,
    bias="none",
    task_type="SEQ_CLS",  # 指定序列分类任务
    fan_in_fan_out=True,
)

5. 调整学习率并补充缺失的导入

LoRA的学习率不宜过高,建议改为1e-4;同时补充必要的导入和compute_metrics函数:

from transformers import AutoModelForSequenceClassification, Trainer, TrainingArguments, DataCollatorWithPadding
import evaluate
import numpy as np

def compute_metrics(eval_pred):
    metric = evaluate.load("accuracy")
    logits, labels = eval_pred
    predictions = np.argmax(logits, axis=-1)
    return metric.compute(predictions=predictions, references=labels)

修复后的完整代码

from datasets import load_dataset
from transformers import AutoTokenizer, AutoModelForSequenceClassification, Trainer, TrainingArguments
import evaluate
import numpy as np
from peft import LoraConfig, get_peft_model

# 加载数据集
splits = ["train", "test"]
ds = {split: ds for split, ds in zip(splits, load_dataset("imdb", split=splits))}

# 初始化tokenizer
tokenizer = AutoTokenizer.from_pretrained("gpt2")
tokenizer.pad_token = tokenizer.eos_token

# 预处理函数
def preprocess_function(examples):
    tokens = tokenizer(examples['text'], padding='max_length', truncation=True, max_length=512)
    tokens["label"] = examples["label"]
    return tokens

# 处理数据集
tokenized_ds = {}
for split in splits:
    tokenized_ds[split] = ds[split].map(preprocess_function, batched=True)

# 加载GPT-2分类模型
model2 = AutoModelForSequenceClassification.from_pretrained(
    "gpt2",
    num_labels=2,
    id2label={0: "NEGATIVE", 1: "POSITIVE"},
    label2id={"NEGATIVE": 0, "POSITIVE": 1},
)
model2.config.pad_token_id = tokenizer.pad_token_id

# 配置LoRA
lora_config = LoraConfig(
    r=8,
    lora_alpha=32,
    target_modules=["c_attn"],
    lora_dropout=0.05,
    bias="none",
    task_type="SEQ_CLS",
    fan_in_fan_out=True,
)
lora_model = get_peft_model(model2, lora_config)

# 定义评估指标
def compute_metrics(eval_pred):
    metric = evaluate.load("accuracy")
    logits, labels = eval_pred
    predictions = np.argmax(logits, axis=-1)
    return metric.compute(predictions=predictions, references=labels)

# 初始化Trainer
trainer_lora = Trainer(
    model=lora_model,
    args=TrainingArguments(
        output_dir="./data/sentiment_analysis2",
        learning_rate=1e-4,  # 调整为LoRA合适的学习率
        per_device_train_batch_size=4,
        per_device_eval_batch_size=4,
        num_train_epochs=5,
        weight_decay=0.01,
        evaluation_strategy="epoch",
        save_strategy="epoch",
        load_best_model_at_end=True,
        logging_dir="./logs",
    ),
    train_dataset=tokenized_ds["train"],
    eval_dataset=tokenized_ds["test"],
    tokenizer=tokenizer,
    compute_metrics=compute_metrics,
)

# 开始训练
trainer_lora.train()

内容的提问来源于stack exchange,提问作者tlanigan

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.29 12:34:59