You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

PyTorch DDP训练脚本仅完成首个epoch后无报错终止问题排查

多GPU训练脚本首个epoch后无报错停止问题

正在移植一个原本在多GPU机器上可正常运行的训练脚本,使用torchrun执行时能检测到全部8个GPU,首个epoch可正常运行,但之后脚本无任何报错信息就停止了。

核心代码片段

LEARNING_RATE = 10e-5  
BATCH_SIZE = 32
BACKEND = "nccl"
os.environ["CUDA_VISIBLE_DEVICES"] = "0,1,2,3,4,5,6,7"
device = torch.device("cuda:0" if torch.cuda.is_available() else "cpu")
print(device)
dist.init_process_group(BACKEND)
local_rank = int(os.environ["LOCAL_RANK"])
torch.cuda.set_device(local_rank)


model = WildfireMetnet(
    forecast_steps=1,
    input_size=64,
    num_input_timesteps=9,
    upsampler_channels=128,
    lstm_channels=32,
    encoder_channels=64,
    center_crop_size=1,
    input_channels=18,
    output_channels=1
).to(local_rank)

model = nn.parallel.DistributedDataParallel(model, device_ids=[local_rank], output_device=local_rank, find_unused_parameters=True)

dataset = WildfireDataset()

train_sampler = DistributedSampler(dataset, shuffle=True)
test_sampler = DistributedSampler(dataset, shuffle=True)
val_sampler = DistributedSampler(dataset, shuffle=True)


train_loader = DataLoader(
    dataset=dataset, 
    batch_size=BATCH_SIZE,
    sampler=train_sampler,
    num_workers=8,
    pin_memory=True
)
test_loader = DataLoader(
    dataset=dataset,  
    batch_size=BATCH_SIZE,
    sampler=test_sampler,
    num_workers=8,
    pin_memory=True
)
val_loader = DataLoader(
    dataset=dataset, 
    batch_size=BATCH_SIZE,
    sampler=val_sampler,
    num_workers=8,
    pin_memory=True
)

optimizer = torch.optim.Adam(model.parameters(), lr=LEARNING_RATE, weight_decay=1e-5) # TODO: tune this, why not the default value of 1e-4?
criterion = torch.nn.BCELoss()

min_val = -torch.inf

            
def train_step(engine: Engine, batch: tuple[torch.Tensor, torch.Tensor]) -> float:
    model.train()
    optimizer.zero_grad()
    features, labels = batch[0].to(local_rank), batch[1].to(local_rank).float()
    features = torch.nan_to_num(features, nan= 0.0)
    labels = torch.nan_to_num(labels, nan= 0.0)
    out = model(features, 0)
    print(out.mean(), labels.mean())
    loss = criterion(out, labels)
    loss.backward()
    optimizer.step()
    return loss.item()

trainer = Engine(train_step)

def validation_step(engine: Engine, batch: tuple[torch.Tensor, torch.Tensor]) -> float:
    model.eval()
    with torch.no_grad():
        features, labels = batch[0].to(local_rank), batch[1].to(local_rank).float()
        features = torch.nan_to_num(features, nan= 0.0)
        labels = torch.nan_to_num(labels, nan= 0.0)
        out = model(features, 0)
        rounded = (out>=0.65).float()
        return rounded, labels

存储指标函数示例

@trainer.on(Events.EPOCH_COMPLETED)
def log_validation_results(trainer):
    evaluator.run(val_loader)
    metrics = evaluator.state.metrics
    filename = 'evaluation/validation_output_2702_multi.txt'
    if not os.path.exists(filename):
        open(filename, 'w').close()
    with open(filename,'a') as f:
        print("{},{:.2f},{:.2f},{:.2f},{:.2f}".format(trainer.state.epoch, metrics["loss"],metrics["precision"],metrics["recall"], metrics["f1"]), file = f)

脚本末尾代码

trainer.run(train_loader, max_epochs=100)
dist.destroy_process_group()

运行环境配置

硬件

  • 8×NVIDIA A100 80GB GPU
  • 双AMD Rome 7742 CPU(128核)
  • 2TB系统内存
  • 30TB NVMe存储

软件

  • DGX-OS 5.4.2(Ubuntu 20.04 LTS)
  • Linux内核5.4.0
  • NVIDIA驱动470.161
  • CUDA 11.4
  • GLIBC 2.31
  • Docker 20.10.21

依赖版本

  • PyTorch 1.13.1
  • Ignite 0.4.10

已尝试操作

  • 调整local_rank配置
  • 将trainer.run的max_epochs设为2,问题仍存在

内容的提问来源于stack exchange,提问作者Nisse97

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.30 13:02:34