You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

PPO智能体接入类DOOM游戏后Pygame窗口无响应问题排查

问题定位与解决:Pygame窗口无响应(PPO智能体整合DOOM类游戏)

问题描述

开发类DOOM的Python游戏并整合PPO智能体后,运行main.py时Pygame窗口无响应,终端无报错回溯,但游戏单独运行正常。相关代码如下:

main.py 相关代码

class Game:
    def getGameState(self):
        currPlayerPos = self.player.pos
        enemyPositions = self.objHandler.npcPositions
        currPlayerHealth = self.player.health

        player_pos_dim = 2
        enemy_pos_dim = 2 * len(enemyPositions)
        player_health_dim = 1
        state_dim = player_pos_dim + enemy_pos_dim + player_health_dim

        action_dim = 3

        state = (
            [currPlayerPos[0], currPlayerPos[1]]
            + [pos for enemyPos in enemyPositions for pos in enemyPos]
            + [currPlayerHealth]
        )

        return state, state_dim, action_dim
    
    def reset(self):
        self.newGame()
        return self.getGameState()
    
    def step(self, action):
        next_state = self.getGameState()
        reward = 0
        done = False
        return next_state, reward, done


if __name__ == "__main__":
    game = Game()
    state, state_dim, action_dim = game.getGameState()

    state_dim = int(state_dim)

    action_dim = 7

    hidden_dim = 64
    agent = PPOAgent(game, state_dim, action_dim, hidden_dim)

    agent.play_game(num_episodes=10)

PPO.py 完整代码

import torch
import torch.nn as nn
import torch.optim as optim
from torch.distributions import Categorical


class ActorCritic(nn.Module):
    def __init__(self, state_dim, action_dim, hidden_dim):
        super(ActorCritic, self).__init__()

        self.actor = nn.Sequential(
            nn.Linear(state_dim, hidden_dim),
            nn.ReLU(),
            nn.Linear(hidden_dim, hidden_dim),
            nn.ReLU(),
            nn.Linear(hidden_dim, action_dim),
            nn.Softmax(dim=-1)
        )

        self.critic = nn.Sequential(
            nn.Linear(state_dim, hidden_dim),
            nn.ReLU(),
            nn.Linear(hidden_dim, hidden_dim),
            nn.ReLU(),
            nn.Linear(hidden_dim, 1)
        )

        self.rewards = []

    def forward(self, state):
        action_probs = self.actor(state)
        value = self.critic(state)
        return action_probs, value


class PPOAgent:
    def __init__(self, game, state_dim, action_dim, hidden_dim):
        self.game = game
        self.state_dim = state_dim
        self.action_dim = action_dim
        self.hidden_dim = hidden_dim
        self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
        self.policy = ActorCritic(state_dim, action_dim, hidden_dim).to(self.device)
        self.optimizer = optim.Adam(self.policy.parameters(), lr=0.001)
        self.gamma = 0.99
        self.epsilon = 0.2
        self.critic_coef = 0.5
        self.entropy_coef = 0.01

    def play_game(self, num_episodes):
        update_interval = 1
        for episode in range(num_episodes):
            state, done = self.game.getGameState(), False
            while not done:
                state_tensor = torch.FloatTensor(state[0]).unsqueeze(0).to(self.device)
                action_probs, value = self.policy(state_tensor)
                dist = Categorical(action_probs)
                action = dist.sample().item()
                next_state, reward, done = self.game.step(action)
                next_state_tensor = torch.FloatTensor(next_state[0]).unsqueeze(0).to(self.device)
                _, next_value = self.policy(next_state_tensor)
                advantage = reward + self.gamma * next_value * (1 - int(done)) - value
                critic_loss = advantage.pow(2).mean()
                prob = action_probs.squeeze(0)[action]
                old_prob = prob.detach()
                ratio = prob / (old_prob + 1e-5)
                surrogate1 = ratio * advantage
                surrogate2 = torch.clamp(ratio, 1 - self.epsilon, 1 + self.epsilon) * advantage
                actor_loss = -torch.min(surrogate1, surrogate2).mean()
                entropy = dist.entropy().mean()
                total_loss = actor_loss + self.critic_coef * critic_loss - self.entropy_coef * entropy
                self.optimizer.zero_grad()
                total_loss.backward()
                self.optimizer.step()
                state = next_state

                if len(self.policy.rewards) >= update_interval or done:
                    self.update_policy()

            print(f"Episode {episode+1}: Total Reward = {total_reward}")

    def update_policy(self):
        R = 0
        G = []

        for reward in self.policy.rewards[::-1]:
            R = reward + gamma * R
            G.insert(0, R)

        G = torch.tensor(G).to(device)
        G = (G - G.mean()) / (G.std() + 1e-9)

        loss = 0
        for log_prob, value, g in zip(
            self.policy.saved_action_probs, self.policy.saved_values, G
        ):
            advantage = g - value.item()

            actor_loss = -log_prob * advantage
            critic_loss = F.smooth_l1_loss(value, torch.tensor([g]).to(device))

            loss += actor_loss + critic_loss

        self.optimizer.zero_grad()
        loss.backward()
        self.optimizer.step()

        self.policy.clear_saved()

问题根源分析

  1. Pygame事件循环阻塞:智能体的训练循环持续占用CPU,没有给Pygame处理窗口事件、刷新画面的时间,导致窗口无响应。
  2. Game类step方法无效:当前step方法仅返回固定状态,未执行任何action对应的游戏逻辑,也不会将done设为True,导致训练循环无限执行。
  3. PPO代码逻辑混乱:
    • play_game中同时执行单步更新和调用update_policy,不符合PPO“收集轨迹后批量更新”的逻辑。
    • update_policy引用未定义变量(gamma、device、F),且ActorCritic类缺少saved_action_probs、saved_values属性及clear_saved方法。
  4. 参数不一致:main中手动将action_dim设为7,但Game类getGameState返回的action_dim为3,导致智能体输出的action无法被游戏正确处理。

解决步骤

1. 修复Pygame窗口阻塞问题

在训练循环的每一步加入Pygame事件处理和窗口刷新逻辑,确保窗口能响应操作:

# 在play_game的while循环内,每次step之后添加
import pygame
# ...
next_state, reward, done = self.game.step(action)
# 新增事件处理
for event in pygame.event.get():
    if event.type == pygame.QUIT:
        done = True
        pygame.quit()
        exit()
# 刷新窗口
pygame.display.flip()
# ...

2. 完善Game类step方法

让step方法真正执行action逻辑、更新游戏状态、计算合理奖励并判断游戏结束:

class Game:
    def __init__(self):
        # 初始化时记录初始生命值
        self.prev_health = self.player.health
        # ...其他初始化逻辑

    def step(self, action):
        # 执行action对应的游戏操作
        if action == 0:
            self.player.move_left()
        elif action == 1:
            self.player.move_right()
        elif action == 2:
            self.player.shoot()
        # 扩展到7个action的话,继续添加对应逻辑(比如前进、后退、换武器等)

        # 更新游戏状态(处理敌人移动、碰撞检测等)
        self.update_game()

        # 计算奖励
        reward = 0
        # 掉血惩罚
        if self.player.health < self.prev_health:
            reward -= 5
            self.prev_health = self.player.health
        # 击杀敌人奖励
        killed_count = self.objHandler.check_killed_enemies()
        reward += killed_count * 10
        # 游戏结束奖励/惩罚
        done = False
        if self.player.health <= 0:
            reward -= 100
            done = True
        elif self.objHandler.all_enemies_dead():
            reward += 100
            done = True

        # 获取新状态
        next_state, _, _ = self.getGameState()
        return next_state, reward, done

3. 修复PPO代码逻辑错误

修正ActorCritic类

添加轨迹存储属性和清空方法:

class ActorCritic(nn.Module):
    def __init__(self, state_dim, action_dim, hidden_dim):
        super(ActorCritic, self).__init__()
        # ...原有网络结构
        self.rewards = []
        self.saved_action_probs = []
        self.saved_values = []

    def forward(self, state):
        # ...原有forward逻辑

    def save_trajectory(self, prob, value, reward):
        self.saved_action_probs.append(prob)
        self.saved_values.append(value)
        self.rewards.append(reward)

    def clear_saved(self):
        self.rewards.clear()
        self.saved_action_probs.clear()
        self.saved_values.clear()

修正PPOAgent的play_game和update_policy方法

改为收集批量轨迹后再更新策略:

import torch.nn.functional as F

class PPOAgent:
    # ...原有__init__逻辑

    def play_game(self, num_episodes):
        update_interval = 10  # 改为每10步或 episode结束后更新
        for episode in range(num_episodes):
            state, _, _ = self.game.reset()
            total_reward = 0
            done = False
            while not done:
                state_tensor = torch.FloatTensor(state).unsqueeze(0).to(self.device)
                action_probs, value = self.policy(state_tensor)
                dist = Categorical(action_probs)
                action = dist.sample().item()
                next_state, reward, done = self.game.step(action)
                total_reward += reward

                # 存储轨迹数据
                self.policy.save_trajectory(
                    dist.log_prob(torch.tensor(action).to(self.device)),
                    value,
                    reward
                )

                # 达到更新间隔或episode结束时更新策略
                if len(self.policy.rewards) >= update_interval or done:
                    self.update_policy()

                state = next_state

            print(f"Episode {episode+1}: Total Reward = {total_reward}")

    def update_policy(self):
        R = 0
        G = []
        # 计算折扣回报
        for reward in reversed(self.policy.rewards):
            R = reward + self.gamma * R
            G.insert(0, R)

        G = torch.tensor(G, dtype=torch.float32).to(self.device)
        G = (G - G.mean()) / (G.std() + 1e-9)

        loss = 0
        for log_prob, value, g in zip(
            self.policy.saved_action_probs, self.policy.saved_values, G
        ):
            advantage = g - value.item()
            # 计算actor和critic损失
            actor_loss = -log_prob * advantage
            critic_loss = F.smooth_l1_loss(value, torch.tensor([g]).to(self.device))
            loss += actor_loss + self.critic_coef * critic_loss

        # 添加熵正则化(可选,提升探索性)
        # 熵计算可以在收集轨迹时存储,这里简化处理
        # entropy = dist.entropy().mean()
        # loss -= self.entropy_coef * entropy

        self.optimizer.zero_grad()
        loss.backward()
        self.optimizer.step()

        self.policy.clear_saved()

4. 统一action_dim参数

要么将Game类中getGameState的action_dim改为7,要么将main中的action_dim改回3,确保智能体输出的action数量与游戏支持的操作一致。

内容的提问来源于stack exchange,提问作者Keanu

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.18 00:05:05