解决GPT-2+LoRA训练IMDb情感分析时的ValueError问题
GPT-2 + LoRA 做IMDb情感分析时出现input_ids缺失错误
我正在基于IMDb数据集和GPT-2模型开发情感分析玩具项目,目标是学习PEFT、LoRA技术并熟悉Huggingface库。以下是我的代码:
from datasets import load_dataset splits = ["train", "test"] ds = {split: ds for split, ds in zip(splits, load_dataset("imdb", split=splits))} from transformers import AutoTokenizer tokenizer = AutoTokenizer.from_pretrained("gpt2") # GPT-2 Tokenizer doesn't have a padding token. tokenizer.pad_token = tokenizer.eos_token def preprocess_function(examples): """Preprocess the imdb dataset by returning tokenized examples.""" tokens = tokenizer(examples['text'],padding='max_length',truncation=True) return tokens tokenized_ds = {} for split in splits: tokenized_ds[split] = ds[split].map(preprocess_function, batched=True) model2 = AutoModelForSequenceClassification.from_pretrained( "gpt2", num_labels=2, id2label={0: "NEGATIVE", 1: "POSITIVE"}, # For converting predictions to strings label2id={"NEGATIVE": 0, "POSITIVE":1}, ) model2.config.pad_token_id = model.config.eos_token_id from peft import LoraConfig from peft import get_peft_model lora_config = LoraConfig("lora_gpt2", fan_in_fan_out=True,) lora_model = get_peft_model(model2, lora_config) trainer_lora = Trainer( model=lora_model, args=TrainingArguments( output_dir="./data/sentiment_analysis2", learning_rate=2e-3, # Reduce the batch size if you don't have enough memory per_device_train_batch_size=4, per_device_eval_batch_size=4, num_train_epochs=5, weight_decay=0.01, evaluation_strategy="epoch", save_strategy="epoch", load_best_model_at_end=True, ), train_dataset=tokenized_ds["train"], eval_dataset=tokenized_ds["test"], tokenizer=tokenizer, data_collator=DataCollatorWithPadding(tokenizer=tokenizer), compute_metrics=compute_metrics, ) trainer_lora.train()
运行后遇到以下错误:
File /opt/conda/lib/python3.10/site-packages/transformers/tokenization_utils_base.py:3018, in PreTrainedTokenizerBase.pad(self, encoded_inputs, padding, max_length, pad_to_multiple_of, return_attention_mask, return_tensors, verbose) 3016 # The model's main input name, usually `input_ids`, has be passed for padding 3017 if self.model_input_names[0] not in encoded_inputs: -> 3018 raise ValueError( 3019 "You should supply an encoding or a list of encodings to this method " 3020 f"that includes {self.model_input_names[0]}, but you provided {list(encoded_inputs.keys())}" 3021 ) 3023 required_input = encoded_inputs[self.model_input_names[0]] 3025 if required_input is None or (isinstance(required_input, Sized) and len(required_input) == 0): ValueError: You should supply an encoding or a list of encodings to this method that includes input_ids, but you provided ['label']
问题原因与修复方案
1. 预处理函数未保留label字段
预处理函数仅返回tokenizer生成的input_ids和attention_mask,丢失了原始数据集的label字段,导致数据加载时字段混乱,最终触发pad方法的错误。
修复: 在预处理函数中保留label字段:
def preprocess_function(examples): """Preprocess the imdb dataset by returning tokenized examples.""" tokens = tokenizer(examples['text'], padding='max_length', truncation=True) # 保留原始标签字段 tokens["label"] = examples["label"] return tokens
2. pad_token_id配置错误
代码中model2.config.pad_token_id = model.config.eos_token_id引用了未定义的model变量,应改为使用已定义的model2或直接用tokenizer的pad_token_id:
model2.config.pad_token_id = tokenizer.pad_token_id
3. 移除冗余的DataCollatorWithPadding
预处理阶段已经使用padding='max_length'完成了padding,无需再用DataCollatorWithPadding重复处理,这会导致数据加载逻辑冲突,直接移除该参数即可。
4. 完善LoRA配置
原LoRAConfig缺少任务类型和关键参数,需补充序列分类任务的必要配置,确保LoRA层正确应用到GPT-2的注意力模块:
lora_config = LoraConfig( r=8, lora_alpha=32, target_modules=["c_attn"], # GPT-2的注意力层 lora_dropout=0.05, bias="none", task_type="SEQ_CLS", # 指定序列分类任务 fan_in_fan_out=True, )
5. 调整学习率并补充缺失的导入
LoRA的学习率不宜过高,建议改为1e-4;同时补充必要的导入和compute_metrics函数:
from transformers import AutoModelForSequenceClassification, Trainer, TrainingArguments, DataCollatorWithPadding import evaluate import numpy as np def compute_metrics(eval_pred): metric = evaluate.load("accuracy") logits, labels = eval_pred predictions = np.argmax(logits, axis=-1) return metric.compute(predictions=predictions, references=labels)
修复后的完整代码
from datasets import load_dataset from transformers import AutoTokenizer, AutoModelForSequenceClassification, Trainer, TrainingArguments import evaluate import numpy as np from peft import LoraConfig, get_peft_model # 加载数据集 splits = ["train", "test"] ds = {split: ds for split, ds in zip(splits, load_dataset("imdb", split=splits))} # 初始化tokenizer tokenizer = AutoTokenizer.from_pretrained("gpt2") tokenizer.pad_token = tokenizer.eos_token # 预处理函数 def preprocess_function(examples): tokens = tokenizer(examples['text'], padding='max_length', truncation=True, max_length=512) tokens["label"] = examples["label"] return tokens # 处理数据集 tokenized_ds = {} for split in splits: tokenized_ds[split] = ds[split].map(preprocess_function, batched=True) # 加载GPT-2分类模型 model2 = AutoModelForSequenceClassification.from_pretrained( "gpt2", num_labels=2, id2label={0: "NEGATIVE", 1: "POSITIVE"}, label2id={"NEGATIVE": 0, "POSITIVE": 1}, ) model2.config.pad_token_id = tokenizer.pad_token_id # 配置LoRA lora_config = LoraConfig( r=8, lora_alpha=32, target_modules=["c_attn"], lora_dropout=0.05, bias="none", task_type="SEQ_CLS", fan_in_fan_out=True, ) lora_model = get_peft_model(model2, lora_config) # 定义评估指标 def compute_metrics(eval_pred): metric = evaluate.load("accuracy") logits, labels = eval_pred predictions = np.argmax(logits, axis=-1) return metric.compute(predictions=predictions, references=labels) # 初始化Trainer trainer_lora = Trainer( model=lora_model, args=TrainingArguments( output_dir="./data/sentiment_analysis2", learning_rate=1e-4, # 调整为LoRA合适的学习率 per_device_train_batch_size=4, per_device_eval_batch_size=4, num_train_epochs=5, weight_decay=0.01, evaluation_strategy="epoch", save_strategy="epoch", load_best_model_at_end=True, logging_dir="./logs", ), train_dataset=tokenized_ds["train"], eval_dataset=tokenized_ds["test"], tokenizer=tokenizer, compute_metrics=compute_metrics, ) # 开始训练 trainer_lora.train()
内容的提问来源于stack exchange,提问作者tlanigan
相关产品推荐
相关产品推荐

