You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

自定义环境DRL优化代码报错求助:mean()参数应为Tensor而非list

解决DRL+PyTorch自定义Gym环境报错:mean(): argument 'input' (position 1) must be Tensor, not list

问题背景

用户为DRL新手,尝试用DRL结合PyTorch求解带约束的简单优化问题,自定义Gym环境时触发如下报错:

Error: mean(): argument 'input' (position 1) must be Tensor, not list

待求解优化问题:

Min  2*X1²+4*X2
subject to X1+5*X2≥5

注:用户明确知晓存在更优优化方法,此案例仅用于演示复杂模型开发思路。

原始代码

import numpy as np
import torch
import torch.nn as nn
import torch.optim as optim
import gym
import matplotlib.pyplot as plt

# Define a custom Gym environment for the constrained optimization problem
class CustomConstrainedEnv(gym.Env):
    def __init__(self):
        super(CustomConstrainedEnv, self).__init__()
        self.action_space = gym.spaces.Box(low=-1, high=1, shape=(2,), dtype=np.float32)
        self.observation_space = gym.spaces.Box(low=-1, high=1, shape=(2,), dtype=np.float32)

    def reset(self):
        self.state = torch.rand(2) * 2 - 1
        return self.state

    def step(self, action):
        self.state = torch.clamp(self.state, -1, 1)
        x1, x2 = self.state
        objective = 2 * x1**2 + 4 * x2
        constraint = x1 + 5 * x2 - 5

        reward = -objective
        penalty = -1e6 * torch.max(torch.tensor([0.0]), constraint)

        done = False
        return self.state, reward + penalty, done, {}

# Define a neural network model for the policy
class PolicyModel(nn.Module):
    def __init__(self, num_actions):
        super(PolicyModel, self).__init__()
        self.fc1 = nn.Linear(2, 64)
        self.fc2 = nn.Linear(64, 64)
        self.mean_head = nn.Linear(64, num_actions)
        self.std_head = nn.Linear(64, num_actions)

    def forward(self, x):
        x = torch.relu(self.fc1(x))
        x = torch.relu(self.fc2(x))
        mean = torch.tanh(self.mean_head(x))
        std = self.softplus(self.std_head(x))  # Use the built-in softplus
        return mean, std

    def softplus(self, x):
        return torch.log(1 + torch.exp(x))

# Hyperparameters
learning_rate = 0.001
gamma = 0.99
num_epochs = 500
num_episodes = 100

# Create the custom environment
env = CustomConstrainedEnv()

# Build the policy model
num_actions = env.action_space.shape[0]
policy_model = PolicyModel(num_actions)
optimizer = optim.Adam(policy_model.parameters(), lr=learning_rate)

# Training loop
reward_history = []

for epoch in range(num_epochs):
    states, actions, rewards, old_means, old_stds, returns, advantages = [], [], [], [], [], [], []

    for episode in range(num_episodes):
        state = env.reset()
        done = False

        #while not done:
        for _ in range(10):
            action_means, action_stds = policy_model(state)
            action = torch.normal(action_means, action_stds)
            action = torch.clamp(action, -1, 1)

            new_state, reward, done, _ = env.step(action)
            old_mean, old_std = action_means, action_stds

            states.append(state)
            actions.append(action)
            rewards.append(reward)
            old_means.append(old_mean)
            old_stds.append(old_std)

            state = new_state

        discounted_reward = 0
        advantage = 0
        for t in reversed(range(len(rewards))):
            discounted_reward = rewards[t] + gamma * discounted_reward
            advantage = discounted_reward - old_means[t]
            returns.insert(0, discounted_reward)
            advantages.insert(0, advantage)

        advantages = (advantages - torch.mean(advantages)) / (torch.std(advantages) + 1e-8)

        policy_loss = []
        for t in range(len(states)):
            action_means, action_stds = policy_model(states[t])
            action_dist = torch.distributions.Normal(action_means, action_stds)
            new_action_probs = action_dist.log_prob(actions[t])
            old_action_probs = action_dist.log_prob(actions[t])
            prob_ratio = torch.exp(new_action_probs - old_action_probs)

            surrogate_loss = torch.min(
                prob_ratio * advantages[t],
                torch.clamp(prob_ratio, 1 - 0.2, 1 + 0.2) * advantages[t]
            )
            policy_loss.append(-surrogate_loss)

        policy_loss = torch.stack(policy_loss).mean()

        optimizer.zero_grad()
        policy_loss.backward()
        optimizer.step()

    # Evaluate the learned policy
    total_rewards = []
    for _ in range(10):
        state = env.reset()
        done = False
        episode_reward = 0

        #while not done:
        for _ in range(10):    
            action_means, _ = policy_model(state)
            action = action_means
            state, reward, done, _ = env.step(action)
            episode_reward += reward

        total_rewards.append(episode_reward)

    avg_reward = np.mean(total_rewards)
    reward_history.append(avg_reward)

    print(f"Epoch {epoch + 1}/{num_epochs}, Average Reward: {avg_reward}")

# Plot the learning progress
plt.plot(reward_history)
plt.xlabel('Epoch')
plt.ylabel('Average Reward')
plt.title('PPO Learning Progress')
plt.show()

错误分析与修复方案

1. 核心错误:列表与张量混淆

报错直接原因是advantages是Tensor组成的列表,而非Tensor张量。需先将列表转换为Tensor再进行标准化操作。

2. 其他逻辑问题修复

  • 折扣回报计算作用域错误:原代码在每个episode后对全局rewards列表反向遍历,导致多episode数据混乱。需为每个episode单独维护轨迹数据,处理后再合并到全局列表。
  • 新旧动作概率计算错误:原代码中旧策略概率用当前模型重新计算,未使用存储的old_means和old_stds,导致PPO的概率比值失效。需用存储的旧参数计算旧动作概率。
  • 环境数据类型不符合Gym规范:原环境reset和step返回Tensor,应改为numpy数组,避免后续兼容性问题。

修复后的完整代码

import numpy as np
import torch
import torch.nn as nn
import torch.optim as optim
import gym
import matplotlib.pyplot as plt

# Define a custom Gym environment for the constrained optimization problem
class CustomConstrainedEnv(gym.Env):
    def __init__(self):
        super(CustomConstrainedEnv, self).__init__()
        self.action_space = gym.spaces.Box(low=-1, high=1, shape=(2,), dtype=np.float32)
        self.observation_space = gym.spaces.Box(low=-1, high=1, shape=(2,), dtype=np.float32)

    def reset(self):
        # 返回numpy数组符合Gym规范
        self.state = np.random.rand(2) * 2 - 1
        return self.state.astype(np.float32)

    def step(self, action):
        # 将动作转换为numpy数组
        action = action.numpy() if isinstance(action, torch.Tensor) else action
        self.state = np.clip(self.state, -1, 1)
        x1, x2 = self.state
        objective = 2 * x1**2 + 4 * x2
        constraint = x1 + 5 * x2 - 5

        reward = -objective
        # 约束违反时施加惩罚
        penalty = -1e6 * max(0.0, constraint)

        done = False
        # 返回numpy数组和数值类型奖励
        return self.state.astype(np.float32), reward + penalty, done, {}

# Define a neural network model for the policy
class PolicyModel(nn.Module):
    def __init__(self, num_actions):
        super(PolicyModel, self).__init__()
        self.fc1 = nn.Linear(2, 64)
        self.fc2 = nn.Linear(64, 64)
        self.mean_head = nn.Linear(64, num_actions)
        self.std_head = nn.Linear(64, num_actions)

    def forward(self, x):
        # 将输入转换为Tensor
        if isinstance(x, np.ndarray):
            x = torch.tensor(x, dtype=torch.float32)
        x = torch.relu(self.fc1(x))
        x = torch.relu(self.fc2(x))
        mean = torch.tanh(self.mean_head(x))
        # 使用PyTorch内置的softplus更稳定
        std = nn.functional.softplus(self.std_head(x))
        return mean, std

# Hyperparameters
learning_rate = 0.001
gamma = 0.99
num_epochs = 500
num_episodes = 100
steps_per_episode = 10

# Create the custom environment
env = CustomConstrainedEnv()

# Build the policy model
num_actions = env.action_space.shape[0]
policy_model = PolicyModel(num_actions)
optimizer = optim.Adam(policy_model.parameters(), lr=learning_rate)

# Training loop
reward_history = []

for epoch in range(num_epochs):
    global_states, global_actions, global_rewards = [], [], []
    global_old_means, global_old_stds = [], []
    global_returns, global_advantages = [], []

    for episode in range(num_episodes):
        state = env.reset()
        episode_states, episode_actions, episode_rewards = [], [], []
        episode_old_means, episode_old_stds = [], []

        for _ in range(steps_per_episode):
            # 获取策略输出
            action_means, action_stds = policy_model(state)
            # 采样动作
            action = torch.normal(action_means, action_stds)
            action = torch.clamp(action, -1, 1)

            # 与环境交互
            new_state, reward, done, _ = env.step(action)

            # 存储单episode轨迹数据
            episode_states.append(torch.tensor(state, dtype=torch.float32))
            episode_actions.append(action)
            episode_rewards.append(torch.tensor(reward, dtype=torch.float32))
            episode_old_means.append(action_means)
            episode_old_stds.append(action_stds)

            state = new_state

        # 计算单episode的折扣回报和优势
        discounted_reward = 0.0
        episode_returns = []
        episode_advantages = []
        # 反向遍历单episode的奖励
        for t in reversed(range(len(episode_rewards))):
            discounted_reward = episode_rewards[t] + gamma * discounted_reward
            episode_returns.insert(0, discounted_reward)
            # 优势计算:回报 - 旧策略均值(简化版优势)
            advantage = discounted_reward - episode_old_means[t].mean()
            episode_advantages.insert(0, advantage)

        # 合并到全局列表
        global_states.extend(episode_states)
        global_actions.extend(episode_actions)
        global_rewards.extend(episode_rewards)
        global_old_means.extend(episode_old_means)
        global_old_stds.extend(episode_old_stds)
        global_returns.extend(episode_returns)
        global_advantages.extend(episode_advantages)

    # 将列表转换为Tensor
    global_advantages = torch.tensor([a.item() for a in global_advantages], dtype=torch.float32)
    # 标准化优势
    global_advantages = (global_advantages - global_advantages.mean()) / (global_advantages.std() + 1e-8)

    # 计算PPO策略损失
    policy_loss = []
    for t in range(len(global_states)):
        # 当前策略输出
        current_mean, current_std = policy_model(global_states[t])
        current_dist = torch.distributions.Normal(current_mean, current_std)
        current_log_prob = current_dist.log_prob(global_actions[t]).sum()

        # 旧策略输出
        old_dist = torch.distributions.Normal(global_old_means[t], global_old_stds[t])
        old_log_prob = old_dist.log_prob(global_actions[t]).sum()

        # 概率比值
        prob_ratio = torch.exp(current_log_prob - old_log_prob)
        # PPO截断损失
        surrogate1 = prob_ratio * global_advantages[t]
        surrogate2 = torch.clamp(prob_ratio, 0.8, 1.2) * global_advantages[t]
        surrogate_loss = torch.min(surrogate1, surrogate2)
        policy_loss.append(-surrogate_loss)

    # 计算平均损失
    policy_loss = torch.stack(policy_loss).mean()

    # 反向传播优化
    optimizer.zero_grad()
    policy_loss.backward()
    optimizer.step()

    # 评估当前策略
    total_rewards = []
    for _ in range(10):
        state = env.reset()
        episode_reward = 0.0
        for _ in range(steps_per_episode):
            action_means, _ = policy_model(state)
            # 直接使用均值作为动作
            action = action_means
            state, reward, done, _ = env.step(action)
            episode_reward += reward
        total_rewards.append(episode_reward)

    avg_reward = np.mean(total_rewards)
    reward_history.append(avg_reward)
    print(f"Epoch {epoch + 1}/{num_epochs}, Average Reward: {avg_reward:.2f}")

# 绘制学习曲线
plt.plot(reward_history)
plt.xlabel('Epoch')
plt.ylabel('Average Reward')
plt.title('PPO Learning Progress')
plt.show()

内容的提问来源于stack exchange,提问作者ali alizadeh

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.06 21:10:56