PyTorch DDP训练脚本仅完成首个epoch后无报错终止问题排查
多GPU训练脚本首个epoch后无报错停止问题
正在移植一个原本在多GPU机器上可正常运行的训练脚本,使用torchrun执行时能检测到全部8个GPU,首个epoch可正常运行,但之后脚本无任何报错信息就停止了。
核心代码片段
LEARNING_RATE = 10e-5 BATCH_SIZE = 32 BACKEND = "nccl" os.environ["CUDA_VISIBLE_DEVICES"] = "0,1,2,3,4,5,6,7" device = torch.device("cuda:0" if torch.cuda.is_available() else "cpu") print(device) dist.init_process_group(BACKEND) local_rank = int(os.environ["LOCAL_RANK"]) torch.cuda.set_device(local_rank) model = WildfireMetnet( forecast_steps=1, input_size=64, num_input_timesteps=9, upsampler_channels=128, lstm_channels=32, encoder_channels=64, center_crop_size=1, input_channels=18, output_channels=1 ).to(local_rank) model = nn.parallel.DistributedDataParallel(model, device_ids=[local_rank], output_device=local_rank, find_unused_parameters=True) dataset = WildfireDataset() train_sampler = DistributedSampler(dataset, shuffle=True) test_sampler = DistributedSampler(dataset, shuffle=True) val_sampler = DistributedSampler(dataset, shuffle=True) train_loader = DataLoader( dataset=dataset, batch_size=BATCH_SIZE, sampler=train_sampler, num_workers=8, pin_memory=True ) test_loader = DataLoader( dataset=dataset, batch_size=BATCH_SIZE, sampler=test_sampler, num_workers=8, pin_memory=True ) val_loader = DataLoader( dataset=dataset, batch_size=BATCH_SIZE, sampler=val_sampler, num_workers=8, pin_memory=True ) optimizer = torch.optim.Adam(model.parameters(), lr=LEARNING_RATE, weight_decay=1e-5) # TODO: tune this, why not the default value of 1e-4? criterion = torch.nn.BCELoss() min_val = -torch.inf def train_step(engine: Engine, batch: tuple[torch.Tensor, torch.Tensor]) -> float: model.train() optimizer.zero_grad() features, labels = batch[0].to(local_rank), batch[1].to(local_rank).float() features = torch.nan_to_num(features, nan= 0.0) labels = torch.nan_to_num(labels, nan= 0.0) out = model(features, 0) print(out.mean(), labels.mean()) loss = criterion(out, labels) loss.backward() optimizer.step() return loss.item() trainer = Engine(train_step) def validation_step(engine: Engine, batch: tuple[torch.Tensor, torch.Tensor]) -> float: model.eval() with torch.no_grad(): features, labels = batch[0].to(local_rank), batch[1].to(local_rank).float() features = torch.nan_to_num(features, nan= 0.0) labels = torch.nan_to_num(labels, nan= 0.0) out = model(features, 0) rounded = (out>=0.65).float() return rounded, labels
存储指标函数示例
@trainer.on(Events.EPOCH_COMPLETED) def log_validation_results(trainer): evaluator.run(val_loader) metrics = evaluator.state.metrics filename = 'evaluation/validation_output_2702_multi.txt' if not os.path.exists(filename): open(filename, 'w').close() with open(filename,'a') as f: print("{},{:.2f},{:.2f},{:.2f},{:.2f}".format(trainer.state.epoch, metrics["loss"],metrics["precision"],metrics["recall"], metrics["f1"]), file = f)
脚本末尾代码
trainer.run(train_loader, max_epochs=100) dist.destroy_process_group()
运行环境配置
硬件
- 8×NVIDIA A100 80GB GPU
- 双AMD Rome 7742 CPU(128核)
- 2TB系统内存
- 30TB NVMe存储
软件
- DGX-OS 5.4.2(Ubuntu 20.04 LTS)
- Linux内核5.4.0
- NVIDIA驱动470.161
- CUDA 11.4
- GLIBC 2.31
- Docker 20.10.21
依赖版本
- PyTorch 1.13.1
- Ignite 0.4.10
已尝试操作
- 调整local_rank配置
- 将
trainer.run的max_epochs设为2,问题仍存在
内容的提问来源于stack exchange,提问作者Nisse97
相关产品推荐
相关产品推荐

