You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

PyTorch中实现自适应权重解决PINN边界条件损失主导问题

解决PINN中BC与PDE损失不平衡的自适应权重方案

问题背景

在一维热传导问题的PINN训练中,直接将边界条件(BC)损失与偏微分方程(PDE)损失相加时,BC损失会主导总损失,导致PDE约束无法有效被模型学习。以下是已实现的完整代码,以及针对该问题的几种自适应权重解决方案。


已实现的基础代码

1. 一维热传导有限差分求解代码

import torch
import torch.nn as nn
import matplotlib.pyplot as plt
import numpy as np
from pyDOE import lhs

######### Finite difference solution
# geometry:
L = 1 # length of the rod
# mesh:
dx = 0.01
nx = int(L/dx) + 1
x = np.linspace(0, L, nx)
# temporal grid:
t_sim = 1
dt = 0.01
nt = int (t_sim/dt)
# parametrization
alpha = 0.14340344168260039
# IC
t_ic = 4
# BC
t_left = 5 # left side with 6 °C temperature
t_right = 3 # right side with 4 °C temperature
# Results
T = np.ones(nx) * t_ic
all_T = []
for i in range (0, nt):
    Tn = T.copy()
    T[1:-1] = Tn[1:-1] + dt/(dx**2) * alpha * (Tn[2:] - 2*Tn[1:-1] + Tn[0:-2])  # 修正原代码中的dx++2错误
    T[0] = t_left
    T[-1] = t_right
    all_T.append(Tn)

2. PINN模型数据准备代码

x = torch.linspace(0, L, nx, dtype=torch.float32)
t = torch.linspace(0, t_sim, nt, dtype=torch.float32)
T_grid, X_grid = torch.meshgrid(t, x)
Temps = np.concatenate(all_T).reshape(nt, nx)
x_test = torch.hstack((X_grid.transpose(1,0).flatten()[:,None], T_grid.transpose(1,0).flatten()[:,None])) 
y_test = torch.from_numpy(Temps) # 真实标签
lb = x_test[0] # 边界下限
ub = x_test[-1] # 边界上限
left_x = torch.hstack((X_grid[:,0][:,None], T_grid[:,0][:,None])) # 左边界的x和t
left_y = torch.ones(left_x.shape[0], 1) * t_left # 左边界温度
left_y[0,0] = t_ic
right_x = torch.hstack((X_grid[:,-1][:,None], T_grid[:,0][:,None]))  # 右边界的x和t
right_y = torch.ones(right_x.shape[0], 1) * t_right # 右边界温度
right_y[0,0] = t_ic
bottom_x = torch.hstack((X_grid[0,1:-1][:,None], T_grid[0,1:-1][:,None])) # 初始条件的x和t
bottom_y = torch.ones(bottom_x.shape[0], 1) * t_ic # 初始条件温度
No_BC = 1 # 使用全部BC数据训练
No_IC = 1 # 使用全部IC数据训练
idx_l = np.random.choice(left_x.shape[0], int(left_x.shape[0]*No_BC), replace=False)
idx_r = np.random.choice(right_x.shape[0], int(right_x.shape[0]*No_BC), replace=False)
idx_b = np.random.choice(bottom_x.shape[0], int(bottom_x.shape[0]*No_IC), replace=False)
X_train_No = torch.vstack([left_x[idx_l,:], right_x[idx_r,:], bottom_x[idx_b,:]])
Y_train_No = torch.vstack([left_y[idx_l,:], right_y[idx_r,:], bottom_y[idx_b,:]])
N_f = 5000
X_train_Nf = lb + (ub-lb)*lhs(2, N_f) 
f_hat = torch.zeros(X_train_Nf.shape[0], 1, dtype=torch.float32) # PDE损失的目标值

3. 基础PINN模型定义代码

device = torch.device("cuda" if torch.cuda.is_available() else "cpu")

class FCN(nn.Module):
    def __init__(self, layers):
        super().__init__()
        self.activation = nn.Tanh()
        self.loss_function = nn.MSELoss(reduction='mean')
        self.linears = nn.ModuleList([nn.Linear(layers[i], layers[i+1]) for i in range(len(layers)-1)]) 
        # Xavier初始化
        for i in range(len(layers)-1):         
            nn.init.xavier_normal_(self.linears[i].weight.data, gain=1.0)            
            nn.init.zeros_(self.linears[i].bias.data)   
    def forward(self, x):
        if not torch.is_tensor(x):         
            x = torch.from_numpy(x)                
        x = x.to(device).float()
        for i in range(len(layers)-2):  
            x = self.activation(self.linears[i](x))    
        x = self.linears[-1](x)
        return x
    # BC损失
    def lossBC(self, x_BC, y_BC):
        pred = self.forward(x_BC)
        return self.loss_function(pred, y_BC.to(device).float())
    # PDE损失
    def lossPDE(self, x_PDE):
        x_PDE = x_PDE.to(device).float()
        g = x_PDE.clone()
        g.requires_grad = True 
        f = self.forward(g)
        # 一阶导数
        f_x_t = torch.autograd.grad(f, g, torch.ones([g.shape[0],1]).to(device), retain_graph=True, create_graph=True)[0]
        # 二阶导数
        f_xx_tt = torch.autograd.grad(f_x_t, g, torch.ones(g.shape).to(device), create_graph=True)[0]
        f_t = f_x_t[:,[1]]
        f_xx = f_xx_tt[:,[0]]
        pde_residual = f_t - alpha * f_xx
        return self.loss_function(pde_residual, f_hat.to(device))

自适应权重解决方案

方案1:动态损失归一化

通过计算两种损失的滑动历史均值,动态缩放损失值,使它们的量级保持一致。修改模型类并调整训练代码:

修改后的模型类

class FCN_Weighted(nn.Module):
    def __init__(self, layers):
        super().__init__()
        self.activation = nn.Tanh()
        self.loss_function = nn.MSELoss(reduction='mean')
        self.linears = nn.ModuleList([nn.Linear(layers[i], layers[i+1]) for i in range(len(layers)-1)]) 
        # Xavier初始化
        for i in range(len(layers)-1):         
            nn.init.xavier_normal_(self.linears[i].weight.data, gain=1.0)            
            nn.init.zeros_(self.linears[i].bias.data)   
        # 损失历史滑动均值(用于归一化)
        self.bc_loss_avg = torch.tensor(1.0, device=device)
        self.pde_loss_avg = torch.tensor(1.0, device=device)
        self.beta = 0.99 # 滑动平均衰减系数
    def forward(self, x):
        if not torch.is_tensor(x):         
            x = torch.from_numpy(x)                
        x = x.to(device).float()
        for i in range(len(layers)-2):  
            x = self.activation(self.linears[i](x))    
        x = self.linears[-1](x)
        return x
    def lossBC(self, x_BC, y_BC):
        pred = self.forward(x_BC)
        return self.loss_function(pred, y_BC.to(device).float())
    def lossPDE(self, x_PDE):
        x_PDE = x_PDE.to(device).float()
        g = x_PDE.clone()
        g.requires_grad = True 
        f = self.forward(g)
        f_x_t = torch.autograd.grad(f, g, torch.ones([g.shape[0],1]).to(device), retain_graph=True, create_graph=True)[0]
        f_xx_tt = torch.autograd.grad(f_x_t, g, torch.ones(g.shape).to(device), create_graph=True)[0]
        f_t = f_x_t[:,[1]]
        f_xx = f_xx_tt[:,[0]]
        pde_residual = f_t - alpha * f_xx
        return self.loss_function(pde_residual, f_hat.to(device))
    def loss(self, x_BC, y_BC, x_PDE):
        loss_bc = self.lossBC(x_BC, y_BC)
        loss_pde = self.lossPDE(x_PDE)
        # 更新滑动平均
        self.bc_loss_avg = self.beta * self.bc_loss_avg + (1 - self.beta) * loss_bc.detach()
        self.pde_loss_avg = self.beta * self.pde_loss_avg + (1 - self.beta) * loss_pde.detach()
        # 归一化损失后相加
        loss_bc_scaled = loss_bc / self.bc_loss_avg
        loss_pde_scaled = loss_pde / self.pde_loss_avg
        return loss_bc_scaled + loss_pde_scaled

对应的训练代码

layers = np.array([2, 50, 50, 50, 50, 50, 1])
PINN = FCN_Weighted(layers).to(device)
optimizer = torch.optim.Adam(PINN.parameters(), lr=0.001)

total_l = np.array([])
BC_l = np.array([])
PDE_l = np.array([])
test_BC_l = np.array([])

for i in range(10000):
    optimizer.zero_grad()
    total_loss = PINN.loss(X_train_No, Y_train_No, X_train_Nf)
    total_loss.backward()
    optimizer.step()
    
    # 记录损失
    with torch.no_grad():
        bc_loss = PINN.lossBC(X_train_No, Y_train_No).cpu().numpy()
        pde_loss = PINN.lossPDE(X_train_Nf).cpu().numpy()
        total_l = np.append(total_l, total_loss.cpu().numpy())
        BC_l = np.append(BC_l, bc_loss)
        PDE_l = np.append(PDE_l, pde_loss)
        test_loss = PINN.lossBC(x_test, y_test.flatten().view(-1,1)).cpu().numpy()
        test_BC_l = np.append(test_BC_l, test_loss)
    
    # 每1000轮打印一次训练状态
    if (i+1) % 1000 == 0:
        print(f"Epoch {i+1}: Total Loss = {total_loss.item():.4f}, BC Loss = {bc_loss:.4f}, PDE Loss = {pde_loss:.4f}")

# 绘制损失曲线
fig,ax=plt.subplots(1,1, figsize=(9,9))
ax.plot(PDE_l, c='g', lw=2, label='PDE loss in train')
ax.plot(BC_l, c='k', lw=2, label='BC loss in train')
ax.plot(test_BC_l, c='r', lw=2, label='BC loss in test')
ax.plot(total_l, c='b', lw=2, label='total loss in train')
ax.set_xlabel('Epoch')
ax.set_ylabel('Loss')
plt.legend()
plt.show()

方案2:可学习权重

将BC和PDE损失的权重设为可学习参数,让模型在训练过程中自动调整权重比例,同时通过指数变换确保权重为正。

修改后的模型类

class FCN_LearnableWeight(nn.Module):
    def __init__(self, layers):
        super().__init__()
        self.activation = nn.Tanh()
        self.loss_function = nn.MSELoss(reduction='mean')
        self.linears = nn.ModuleList([nn.Linear(layers[i], layers[i+1]) for i in range(len(layers)-1)]) 
        # Xavier初始化
        for i in range(len(layers)-1):         
            nn.init.xavier_normal_(self.linears[i].weight.data, gain=1.0)            
            nn.init.zeros_(self.linears[i].bias.data)   
        # 可学习权重,初始值设为1.0
        self.w_bc = nn.Parameter(torch.tensor(1.0, device=device))
        self.w_pde = nn.Parameter(torch.tensor(1.0, device=device))
    def forward(self, x):
        if not torch.is_tensor(x):         
            x = torch.from_numpy(x)                
        x = x.to(device).float()
        for i in range(len(layers)-2):  
            x = self.activation(self.linears[i](x))    
        x = self.linears[-1](x)
        return x
    def lossBC(self, x_BC, y_BC):
        pred = self.forward(x_BC)
        return self.loss_function(pred, y_BC.to(device).float())
    def lossPDE(self, x_PDE):
        x_PDE = x_PDE.to(device).float()
        g = x_PDE.clone()
        g.requires_grad = True 
        f = self.forward(g)
        f_x_t = torch.autograd.grad(f, g, torch.ones([g.shape[0],1]).to(device), retain_graph=True, create_graph=True)[0]
        f_xx_tt = torch.autograd.grad(f_x_t, g, torch.ones(g.shape).to(device), create_graph=True)[0]
        f_t = f_x_t[:,[1]]
        f_xx = f_xx_tt[:,[0]]
        pde_residual = f_t - alpha * f_xx
        return self.loss_function(pde_residual, f_hat.to(device))
    def loss(self, x_BC, y_BC, x_PDE):
        loss_bc = self.lossBC(x_BC, y_BC)
        loss_pde = self.lossPDE(x_PDE)
        # 指数变换确保权重为正,同时将权重加入损失避免其无限增大
        return torch.exp(-self.w_bc) * loss_bc + torch.exp(-self.w_pde) * loss_pde + self.w_bc + self.w_pde

训练代码同方案1,只需替换模型类即可。

方案3:梯度均衡

通过平衡两种损失的梯度贡献,避免某一种损失的梯度主导优化过程。核心是根据两种损失的梯度范数反向分配权重:

修改后的损失计算方法(替换原模型类中的loss方法)

def loss(self, x_BC, y_BC, x_PDE):
    loss_bc = self.lossBC(x_BC, y_BC)
    loss_pde = self.lossPDE(x_PDE)
    
    # 计算BC损失的梯度范数
    grad_bc = torch.autograd.grad(loss_bc, self.parameters(), retain_graph=True, create_graph=True)
    grad_bc_norm = torch.norm(torch.cat([g.flatten() for g in grad_bc]))
    
    # 计算PDE损失的梯度范数
    grad_pde = torch.autograd.grad(loss_pde, self.parameters(), create_graph=True)
    grad_pde_norm = torch.norm(torch.cat([g.flatten() for g in grad_pde]))
    
    # 按梯度范数反向分配权重,均衡梯度贡献
    weight_bc = grad_pde_norm / (grad_bc_norm + grad_pde_norm + 1e-8)
    weight_pde = grad_bc_norm / (grad_bc_norm + grad_pde_norm + 1e-8)
    
    return weight_bc * loss_bc + weight_pde * loss_pde

内容的提问来源于stack exchange,提问作者Link_tester

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.31 15:01:12