PyTorch中实现自适应权重解决PINN边界条件损失主导问题
解决PINN中BC与PDE损失不平衡的自适应权重方案
问题背景
在一维热传导问题的PINN训练中,直接将边界条件(BC)损失与偏微分方程(PDE)损失相加时,BC损失会主导总损失,导致PDE约束无法有效被模型学习。以下是已实现的完整代码,以及针对该问题的几种自适应权重解决方案。
已实现的基础代码
1. 一维热传导有限差分求解代码
import torch import torch.nn as nn import matplotlib.pyplot as plt import numpy as np from pyDOE import lhs ######### Finite difference solution # geometry: L = 1 # length of the rod # mesh: dx = 0.01 nx = int(L/dx) + 1 x = np.linspace(0, L, nx) # temporal grid: t_sim = 1 dt = 0.01 nt = int (t_sim/dt) # parametrization alpha = 0.14340344168260039 # IC t_ic = 4 # BC t_left = 5 # left side with 6 °C temperature t_right = 3 # right side with 4 °C temperature # Results T = np.ones(nx) * t_ic all_T = [] for i in range (0, nt): Tn = T.copy() T[1:-1] = Tn[1:-1] + dt/(dx**2) * alpha * (Tn[2:] - 2*Tn[1:-1] + Tn[0:-2]) # 修正原代码中的dx++2错误 T[0] = t_left T[-1] = t_right all_T.append(Tn)
2. PINN模型数据准备代码
x = torch.linspace(0, L, nx, dtype=torch.float32) t = torch.linspace(0, t_sim, nt, dtype=torch.float32) T_grid, X_grid = torch.meshgrid(t, x) Temps = np.concatenate(all_T).reshape(nt, nx) x_test = torch.hstack((X_grid.transpose(1,0).flatten()[:,None], T_grid.transpose(1,0).flatten()[:,None])) y_test = torch.from_numpy(Temps) # 真实标签 lb = x_test[0] # 边界下限 ub = x_test[-1] # 边界上限 left_x = torch.hstack((X_grid[:,0][:,None], T_grid[:,0][:,None])) # 左边界的x和t left_y = torch.ones(left_x.shape[0], 1) * t_left # 左边界温度 left_y[0,0] = t_ic right_x = torch.hstack((X_grid[:,-1][:,None], T_grid[:,0][:,None])) # 右边界的x和t right_y = torch.ones(right_x.shape[0], 1) * t_right # 右边界温度 right_y[0,0] = t_ic bottom_x = torch.hstack((X_grid[0,1:-1][:,None], T_grid[0,1:-1][:,None])) # 初始条件的x和t bottom_y = torch.ones(bottom_x.shape[0], 1) * t_ic # 初始条件温度 No_BC = 1 # 使用全部BC数据训练 No_IC = 1 # 使用全部IC数据训练 idx_l = np.random.choice(left_x.shape[0], int(left_x.shape[0]*No_BC), replace=False) idx_r = np.random.choice(right_x.shape[0], int(right_x.shape[0]*No_BC), replace=False) idx_b = np.random.choice(bottom_x.shape[0], int(bottom_x.shape[0]*No_IC), replace=False) X_train_No = torch.vstack([left_x[idx_l,:], right_x[idx_r,:], bottom_x[idx_b,:]]) Y_train_No = torch.vstack([left_y[idx_l,:], right_y[idx_r,:], bottom_y[idx_b,:]]) N_f = 5000 X_train_Nf = lb + (ub-lb)*lhs(2, N_f) f_hat = torch.zeros(X_train_Nf.shape[0], 1, dtype=torch.float32) # PDE损失的目标值
3. 基础PINN模型定义代码
device = torch.device("cuda" if torch.cuda.is_available() else "cpu") class FCN(nn.Module): def __init__(self, layers): super().__init__() self.activation = nn.Tanh() self.loss_function = nn.MSELoss(reduction='mean') self.linears = nn.ModuleList([nn.Linear(layers[i], layers[i+1]) for i in range(len(layers)-1)]) # Xavier初始化 for i in range(len(layers)-1): nn.init.xavier_normal_(self.linears[i].weight.data, gain=1.0) nn.init.zeros_(self.linears[i].bias.data) def forward(self, x): if not torch.is_tensor(x): x = torch.from_numpy(x) x = x.to(device).float() for i in range(len(layers)-2): x = self.activation(self.linears[i](x)) x = self.linears[-1](x) return x # BC损失 def lossBC(self, x_BC, y_BC): pred = self.forward(x_BC) return self.loss_function(pred, y_BC.to(device).float()) # PDE损失 def lossPDE(self, x_PDE): x_PDE = x_PDE.to(device).float() g = x_PDE.clone() g.requires_grad = True f = self.forward(g) # 一阶导数 f_x_t = torch.autograd.grad(f, g, torch.ones([g.shape[0],1]).to(device), retain_graph=True, create_graph=True)[0] # 二阶导数 f_xx_tt = torch.autograd.grad(f_x_t, g, torch.ones(g.shape).to(device), create_graph=True)[0] f_t = f_x_t[:,[1]] f_xx = f_xx_tt[:,[0]] pde_residual = f_t - alpha * f_xx return self.loss_function(pde_residual, f_hat.to(device))
自适应权重解决方案
方案1:动态损失归一化
通过计算两种损失的滑动历史均值,动态缩放损失值,使它们的量级保持一致。修改模型类并调整训练代码:
修改后的模型类
class FCN_Weighted(nn.Module): def __init__(self, layers): super().__init__() self.activation = nn.Tanh() self.loss_function = nn.MSELoss(reduction='mean') self.linears = nn.ModuleList([nn.Linear(layers[i], layers[i+1]) for i in range(len(layers)-1)]) # Xavier初始化 for i in range(len(layers)-1): nn.init.xavier_normal_(self.linears[i].weight.data, gain=1.0) nn.init.zeros_(self.linears[i].bias.data) # 损失历史滑动均值(用于归一化) self.bc_loss_avg = torch.tensor(1.0, device=device) self.pde_loss_avg = torch.tensor(1.0, device=device) self.beta = 0.99 # 滑动平均衰减系数 def forward(self, x): if not torch.is_tensor(x): x = torch.from_numpy(x) x = x.to(device).float() for i in range(len(layers)-2): x = self.activation(self.linears[i](x)) x = self.linears[-1](x) return x def lossBC(self, x_BC, y_BC): pred = self.forward(x_BC) return self.loss_function(pred, y_BC.to(device).float()) def lossPDE(self, x_PDE): x_PDE = x_PDE.to(device).float() g = x_PDE.clone() g.requires_grad = True f = self.forward(g) f_x_t = torch.autograd.grad(f, g, torch.ones([g.shape[0],1]).to(device), retain_graph=True, create_graph=True)[0] f_xx_tt = torch.autograd.grad(f_x_t, g, torch.ones(g.shape).to(device), create_graph=True)[0] f_t = f_x_t[:,[1]] f_xx = f_xx_tt[:,[0]] pde_residual = f_t - alpha * f_xx return self.loss_function(pde_residual, f_hat.to(device)) def loss(self, x_BC, y_BC, x_PDE): loss_bc = self.lossBC(x_BC, y_BC) loss_pde = self.lossPDE(x_PDE) # 更新滑动平均 self.bc_loss_avg = self.beta * self.bc_loss_avg + (1 - self.beta) * loss_bc.detach() self.pde_loss_avg = self.beta * self.pde_loss_avg + (1 - self.beta) * loss_pde.detach() # 归一化损失后相加 loss_bc_scaled = loss_bc / self.bc_loss_avg loss_pde_scaled = loss_pde / self.pde_loss_avg return loss_bc_scaled + loss_pde_scaled
对应的训练代码
layers = np.array([2, 50, 50, 50, 50, 50, 1]) PINN = FCN_Weighted(layers).to(device) optimizer = torch.optim.Adam(PINN.parameters(), lr=0.001) total_l = np.array([]) BC_l = np.array([]) PDE_l = np.array([]) test_BC_l = np.array([]) for i in range(10000): optimizer.zero_grad() total_loss = PINN.loss(X_train_No, Y_train_No, X_train_Nf) total_loss.backward() optimizer.step() # 记录损失 with torch.no_grad(): bc_loss = PINN.lossBC(X_train_No, Y_train_No).cpu().numpy() pde_loss = PINN.lossPDE(X_train_Nf).cpu().numpy() total_l = np.append(total_l, total_loss.cpu().numpy()) BC_l = np.append(BC_l, bc_loss) PDE_l = np.append(PDE_l, pde_loss) test_loss = PINN.lossBC(x_test, y_test.flatten().view(-1,1)).cpu().numpy() test_BC_l = np.append(test_BC_l, test_loss) # 每1000轮打印一次训练状态 if (i+1) % 1000 == 0: print(f"Epoch {i+1}: Total Loss = {total_loss.item():.4f}, BC Loss = {bc_loss:.4f}, PDE Loss = {pde_loss:.4f}") # 绘制损失曲线 fig,ax=plt.subplots(1,1, figsize=(9,9)) ax.plot(PDE_l, c='g', lw=2, label='PDE loss in train') ax.plot(BC_l, c='k', lw=2, label='BC loss in train') ax.plot(test_BC_l, c='r', lw=2, label='BC loss in test') ax.plot(total_l, c='b', lw=2, label='total loss in train') ax.set_xlabel('Epoch') ax.set_ylabel('Loss') plt.legend() plt.show()
方案2:可学习权重
将BC和PDE损失的权重设为可学习参数,让模型在训练过程中自动调整权重比例,同时通过指数变换确保权重为正。
修改后的模型类
class FCN_LearnableWeight(nn.Module): def __init__(self, layers): super().__init__() self.activation = nn.Tanh() self.loss_function = nn.MSELoss(reduction='mean') self.linears = nn.ModuleList([nn.Linear(layers[i], layers[i+1]) for i in range(len(layers)-1)]) # Xavier初始化 for i in range(len(layers)-1): nn.init.xavier_normal_(self.linears[i].weight.data, gain=1.0) nn.init.zeros_(self.linears[i].bias.data) # 可学习权重,初始值设为1.0 self.w_bc = nn.Parameter(torch.tensor(1.0, device=device)) self.w_pde = nn.Parameter(torch.tensor(1.0, device=device)) def forward(self, x): if not torch.is_tensor(x): x = torch.from_numpy(x) x = x.to(device).float() for i in range(len(layers)-2): x = self.activation(self.linears[i](x)) x = self.linears[-1](x) return x def lossBC(self, x_BC, y_BC): pred = self.forward(x_BC) return self.loss_function(pred, y_BC.to(device).float()) def lossPDE(self, x_PDE): x_PDE = x_PDE.to(device).float() g = x_PDE.clone() g.requires_grad = True f = self.forward(g) f_x_t = torch.autograd.grad(f, g, torch.ones([g.shape[0],1]).to(device), retain_graph=True, create_graph=True)[0] f_xx_tt = torch.autograd.grad(f_x_t, g, torch.ones(g.shape).to(device), create_graph=True)[0] f_t = f_x_t[:,[1]] f_xx = f_xx_tt[:,[0]] pde_residual = f_t - alpha * f_xx return self.loss_function(pde_residual, f_hat.to(device)) def loss(self, x_BC, y_BC, x_PDE): loss_bc = self.lossBC(x_BC, y_BC) loss_pde = self.lossPDE(x_PDE) # 指数变换确保权重为正,同时将权重加入损失避免其无限增大 return torch.exp(-self.w_bc) * loss_bc + torch.exp(-self.w_pde) * loss_pde + self.w_bc + self.w_pde
训练代码同方案1,只需替换模型类即可。
方案3:梯度均衡
通过平衡两种损失的梯度贡献,避免某一种损失的梯度主导优化过程。核心是根据两种损失的梯度范数反向分配权重:
修改后的损失计算方法(替换原模型类中的loss方法)
def loss(self, x_BC, y_BC, x_PDE): loss_bc = self.lossBC(x_BC, y_BC) loss_pde = self.lossPDE(x_PDE) # 计算BC损失的梯度范数 grad_bc = torch.autograd.grad(loss_bc, self.parameters(), retain_graph=True, create_graph=True) grad_bc_norm = torch.norm(torch.cat([g.flatten() for g in grad_bc])) # 计算PDE损失的梯度范数 grad_pde = torch.autograd.grad(loss_pde, self.parameters(), create_graph=True) grad_pde_norm = torch.norm(torch.cat([g.flatten() for g in grad_pde])) # 按梯度范数反向分配权重,均衡梯度贡献 weight_bc = grad_pde_norm / (grad_bc_norm + grad_pde_norm + 1e-8) weight_pde = grad_bc_norm / (grad_bc_norm + grad_pde_norm + 1e-8) return weight_bc * loss_bc + weight_pde * loss_pde
内容的提问来源于stack exchange,提问作者Link_tester
相关产品推荐
相关产品推荐

