You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Hugging Face微调GPT-2时Segmentation Fault问题排查

微调GPT-2时Segmentation Fault(段错误)的排查与解决

问题描述

使用Hugging Face Transformers库微调GPT-2模型时,训练过程中触发Segmentation fault (core dumped)错误,且无堆栈跟踪信息。可复现的最小代码如下:

import torch
from transformers import GPT2Tokenizer, GPT2LMHeadModel, Trainer, TrainingArguments, DataCollatorForLanguageModeling
from datasets import load_dataset
from typing import Dict
import os

# Check if CUDA is available
device = torch.device("cuda" if torch.cuda.is_available() else "cpu")

# Load pre-trained model and tokenizer
model_name = "gpt2"
tokenizer = GPT2Tokenizer.from_pretrained(model_name)

# Set the pad_token to the eos_token
tokenizer.pad_token = tokenizer.eos_token

model = GPT2LMHeadModel.from_pretrained(model_name)

# Move model to GPU and enable bf16 precision
model = model.to(device=device, dtype=torch.bfloat16)

def preprocess_function_proofnet_simple(examples: Dict[str, list], tokenizer: GPT2Tokenizer, max_length: int = 512) -> Dict[str, torch.Tensor]:
    """
    Preprocess the input data for the proofnet dataset.

    Args:
    examples: The examples to preprocess.
    tokenizer: The tokenizer for encoding the texts.

    Returns:
    The processed model inputs.
    """
    inputs = [f"{examples['nl_statement'][i]}{tokenizer.eos_token}{examples['formal_statement'][i]}" for i in range(len(examples['nl_statement']))]
    model_inputs = tokenizer(inputs, max_length=max_length, padding="max_length", truncation=True, return_tensors="pt")
    labels = model_inputs.input_ids.clone()
    labels[labels == tokenizer.pad_token_id] = -100
    model_inputs["labels"] = labels
    return model_inputs

# Load the dataset
dataset_path = "hoskinson-center/proofnet"
dataset = load_dataset(dataset_path)

# Select only 10 examples for training and validation
small_train_dataset = dataset['validation'].select(range(10))
small_val_dataset = dataset['test'].select(range(10))

# Preprocess the dataset
train_dataset = small_train_dataset.map(lambda examples: preprocess_function_proofnet_simple(examples, tokenizer), batched=True, remove_columns=["nl_statement", "formal_statement"])
val_dataset = small_val_dataset.map(lambda examples: preprocess_function_proofnet_simple(examples, tokenizer), batched=True, remove_columns=["nl_statement", "formal_statement"])

# Data collator for language modeling
data_collator = DataCollatorForLanguageModeling(tokenizer=tokenizer, mlm=False)

# # Training arguments (works fine)
# training_args = TrainingArguments(
#     output_dir=os.path.expanduser("~/tmp/gpt2_trainer"),
#     overwrite_output_dir=True,
#     num_train_epochs=3,  # Train for 3 epochs
#     per_device_train_batch_size=2,
#     save_steps=10_000,
#     save_total_limit=2,
#     bf16=True,  # Enable bf16 training only
#     logging_dir=os.path.expanduser("~/tmp/gpt2_trainer/logs"),
#     logging_steps=200,
#     report_to="none"  # Disable logging to WandB
# )

# Training arguments (causes Segmentation Fault)
from pathlib import Path
output_dir_train: Path = Path('~/tmp').expanduser()
output_dir_train.mkdir(parents=True, exist_ok=True)
training_args = TrainingArguments(
    output_dir=output_dir_train,
    max_steps=2,  # TODO get rid of this in favour of 1 or 2 or 3 epochs
    # num_train_epochs=num_train_epochs, 
    gradient_accumulation_steps=2,  # based on alpaca https://github.com/tatsu-lab/stanford_alpaca, allows to process effective_batch_size = gradient_accumulation_steps * batch_size, num its to accumulate before opt update step
    gradient_checkpointing = True,  # TODO depending on hardware set to true?
    per_device_train_batch_size=2,
    per_device_eval_batch_size=2,
    learning_rate=1e-5,
    weight_decay=0.01, 
    max_grad_norm=1.0, # TODO once real training change?
    lr_scheduler_type='cosine',  # TODO once real training change? using what I've seen most in vision 
    warmup_ratio=0.01,
    optim='paged_adamw_32bit',
    # logging_strategy='epoch', # TODO
    save_steps=100, # Save checkpoint every 500 steps
    save_total_limit=3, # save last 3
    logging_steps=10,  # Frequency of logging steps
    logging_first_step=True,
    logging_dir=output_dir_train,
    eval_strategy='no',  # "no"`: No evaluation is done during training. no can be good to avoid memory issues.
    report_to='none',  # options I recommend: 'none', 'wandb'
    fp16=False,  # never ever set to True
    bf16=torch.cuda.is_bf16_supported(),
    # full_determinism=True,  # TODO periphery, Ensure reproducibility
    # torchdynamo="nvfuser",  # TODO periphery, Use NVFuser backend for optimized torch operations
    # dataloader_prefetch_factor=2,  # TODO periphery, Number of batches to prefetch
    # dataloader_pin_memory=True,  # TODO periphery, Pin memory in data loaders for faster transfer to GPU
    # dataloader_num_workers=16,  # TODO Number of subprocesses for data loading
)

# Initialize the Trainer
trainer = Trainer(
    model=model,
    args=training_args,
    data_collator=data_collator,
    train_dataset=train_dataset,
    eval_dataset=val_dataset,
)

# Train the model
trainer.train()

# Save the model
model.save_pretrained(os.path.expanduser("~/tmp/gpt2_trainer/final_model"))
tokenizer.save_pretrained(os.path.expanduser("~/tmp/gpt2_trainer/final_model"))

print('Done!\a')

注释掉的旧训练参数配置可正常运行,新配置触发段错误。

可能原因

对比新旧配置的差异,触发段错误的核心因素是以下几点的组合:

  1. 梯度检查点(gradient_checkpointing=True):该特性通过牺牲计算量节省内存,但在bf16精度下,部分模型的反向传播逻辑可能存在内存访问越界问题。
  2. 32位分页优化器(optim='paged_adamw_32bit'):该优化器的内存分页机制与bf16模型的张量存储格式不兼容,导致底层内存操作出错。
  3. 手动提前转换模型精度:代码中手动执行model = model.to(device=device, dtype=torch.bfloat16),与TrainingArguments中bf16=torch.cuda.is_bf16_supported()的自动精度控制逻辑冲突,引发张量状态异常。

解决方法

针对上述原因,逐一调整配置即可解决问题:

1. 关闭梯度检查点

将gradient_checkpointing设为False,避免内存访问异常:

gradient_checkpointing = False,

2. 更换优化器

使用默认的PyTorch AdamW优化器替代32位分页版本:

optim='adamw_torch',

3. 移除手动精度转换

删除手动将模型转成bf16的代码,让Trainer通过TrainingArguments自动处理精度设置:

# 移除这一行:model = model.to(device=device, dtype=torch.bfloat16)
model = model.to(device=device)

4. 验证硬件bf16支持

如果GPU不支持bf16(如RTX20系列及更早型号),直接关闭bf16:

bf16=False,

验证修改

调整后的核心配置示例:

training_args = TrainingArguments(
    output_dir=output_dir_train,
    max_steps=2,
    gradient_accumulation_steps=2,
    gradient_checkpointing = False,  # 关闭梯度检查点
    per_device_train_batch_size=2,
    per_device_eval_batch_size=2,
    learning_rate=1e-5,
    weight_decay=0.01, 
    max_grad_norm=1.0,
    lr_scheduler_type='cosine',
    warmup_ratio=0.01,
    optim='adamw_torch',  # 使用默认优化器
    save_steps=100,
    save_total_limit=3,
    logging_steps=10,
    logging_first_step=True,
    logging_dir=output_dir_train,
    eval_strategy='no',
    report_to='none',
    fp16=False,
    bf16=torch.cuda.is_bf16_supported(),
)

# 仅移动模型到设备,不手动设置精度
model = model.to(device=device)

运行修改后的代码,即可避免Segmentation Fault错误。

内容的提问来源于stack exchange,提问作者Charlie Parker

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.19 21:23:10