PPO智能体接入类DOOM游戏后Pygame窗口无响应问题排查
问题定位与解决:Pygame窗口无响应(PPO智能体整合DOOM类游戏)
问题描述
开发类DOOM的Python游戏并整合PPO智能体后,运行main.py时Pygame窗口无响应,终端无报错回溯,但游戏单独运行正常。相关代码如下:
main.py 相关代码
class Game: def getGameState(self): currPlayerPos = self.player.pos enemyPositions = self.objHandler.npcPositions currPlayerHealth = self.player.health player_pos_dim = 2 enemy_pos_dim = 2 * len(enemyPositions) player_health_dim = 1 state_dim = player_pos_dim + enemy_pos_dim + player_health_dim action_dim = 3 state = ( [currPlayerPos[0], currPlayerPos[1]] + [pos for enemyPos in enemyPositions for pos in enemyPos] + [currPlayerHealth] ) return state, state_dim, action_dim def reset(self): self.newGame() return self.getGameState() def step(self, action): next_state = self.getGameState() reward = 0 done = False return next_state, reward, done if __name__ == "__main__": game = Game() state, state_dim, action_dim = game.getGameState() state_dim = int(state_dim) action_dim = 7 hidden_dim = 64 agent = PPOAgent(game, state_dim, action_dim, hidden_dim) agent.play_game(num_episodes=10)
PPO.py 完整代码
import torch import torch.nn as nn import torch.optim as optim from torch.distributions import Categorical class ActorCritic(nn.Module): def __init__(self, state_dim, action_dim, hidden_dim): super(ActorCritic, self).__init__() self.actor = nn.Sequential( nn.Linear(state_dim, hidden_dim), nn.ReLU(), nn.Linear(hidden_dim, hidden_dim), nn.ReLU(), nn.Linear(hidden_dim, action_dim), nn.Softmax(dim=-1) ) self.critic = nn.Sequential( nn.Linear(state_dim, hidden_dim), nn.ReLU(), nn.Linear(hidden_dim, hidden_dim), nn.ReLU(), nn.Linear(hidden_dim, 1) ) self.rewards = [] def forward(self, state): action_probs = self.actor(state) value = self.critic(state) return action_probs, value class PPOAgent: def __init__(self, game, state_dim, action_dim, hidden_dim): self.game = game self.state_dim = state_dim self.action_dim = action_dim self.hidden_dim = hidden_dim self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") self.policy = ActorCritic(state_dim, action_dim, hidden_dim).to(self.device) self.optimizer = optim.Adam(self.policy.parameters(), lr=0.001) self.gamma = 0.99 self.epsilon = 0.2 self.critic_coef = 0.5 self.entropy_coef = 0.01 def play_game(self, num_episodes): update_interval = 1 for episode in range(num_episodes): state, done = self.game.getGameState(), False while not done: state_tensor = torch.FloatTensor(state[0]).unsqueeze(0).to(self.device) action_probs, value = self.policy(state_tensor) dist = Categorical(action_probs) action = dist.sample().item() next_state, reward, done = self.game.step(action) next_state_tensor = torch.FloatTensor(next_state[0]).unsqueeze(0).to(self.device) _, next_value = self.policy(next_state_tensor) advantage = reward + self.gamma * next_value * (1 - int(done)) - value critic_loss = advantage.pow(2).mean() prob = action_probs.squeeze(0)[action] old_prob = prob.detach() ratio = prob / (old_prob + 1e-5) surrogate1 = ratio * advantage surrogate2 = torch.clamp(ratio, 1 - self.epsilon, 1 + self.epsilon) * advantage actor_loss = -torch.min(surrogate1, surrogate2).mean() entropy = dist.entropy().mean() total_loss = actor_loss + self.critic_coef * critic_loss - self.entropy_coef * entropy self.optimizer.zero_grad() total_loss.backward() self.optimizer.step() state = next_state if len(self.policy.rewards) >= update_interval or done: self.update_policy() print(f"Episode {episode+1}: Total Reward = {total_reward}") def update_policy(self): R = 0 G = [] for reward in self.policy.rewards[::-1]: R = reward + gamma * R G.insert(0, R) G = torch.tensor(G).to(device) G = (G - G.mean()) / (G.std() + 1e-9) loss = 0 for log_prob, value, g in zip( self.policy.saved_action_probs, self.policy.saved_values, G ): advantage = g - value.item() actor_loss = -log_prob * advantage critic_loss = F.smooth_l1_loss(value, torch.tensor([g]).to(device)) loss += actor_loss + critic_loss self.optimizer.zero_grad() loss.backward() self.optimizer.step() self.policy.clear_saved()
问题根源分析
- Pygame事件循环阻塞:智能体的训练循环持续占用CPU,没有给Pygame处理窗口事件、刷新画面的时间,导致窗口无响应。
- Game类step方法无效:当前step方法仅返回固定状态,未执行任何action对应的游戏逻辑,也不会将
done设为True,导致训练循环无限执行。 - PPO代码逻辑混乱:
play_game中同时执行单步更新和调用update_policy,不符合PPO“收集轨迹后批量更新”的逻辑。update_policy引用未定义变量(gamma、device、F),且ActorCritic类缺少saved_action_probs、saved_values属性及clear_saved方法。
- 参数不一致:main中手动将
action_dim设为7,但Game类getGameState返回的action_dim为3,导致智能体输出的action无法被游戏正确处理。
解决步骤
1. 修复Pygame窗口阻塞问题
在训练循环的每一步加入Pygame事件处理和窗口刷新逻辑,确保窗口能响应操作:
# 在play_game的while循环内,每次step之后添加 import pygame # ... next_state, reward, done = self.game.step(action) # 新增事件处理 for event in pygame.event.get(): if event.type == pygame.QUIT: done = True pygame.quit() exit() # 刷新窗口 pygame.display.flip() # ...
2. 完善Game类step方法
让step方法真正执行action逻辑、更新游戏状态、计算合理奖励并判断游戏结束:
class Game: def __init__(self): # 初始化时记录初始生命值 self.prev_health = self.player.health # ...其他初始化逻辑 def step(self, action): # 执行action对应的游戏操作 if action == 0: self.player.move_left() elif action == 1: self.player.move_right() elif action == 2: self.player.shoot() # 扩展到7个action的话,继续添加对应逻辑(比如前进、后退、换武器等) # 更新游戏状态(处理敌人移动、碰撞检测等) self.update_game() # 计算奖励 reward = 0 # 掉血惩罚 if self.player.health < self.prev_health: reward -= 5 self.prev_health = self.player.health # 击杀敌人奖励 killed_count = self.objHandler.check_killed_enemies() reward += killed_count * 10 # 游戏结束奖励/惩罚 done = False if self.player.health <= 0: reward -= 100 done = True elif self.objHandler.all_enemies_dead(): reward += 100 done = True # 获取新状态 next_state, _, _ = self.getGameState() return next_state, reward, done
3. 修复PPO代码逻辑错误
修正ActorCritic类
添加轨迹存储属性和清空方法:
class ActorCritic(nn.Module): def __init__(self, state_dim, action_dim, hidden_dim): super(ActorCritic, self).__init__() # ...原有网络结构 self.rewards = [] self.saved_action_probs = [] self.saved_values = [] def forward(self, state): # ...原有forward逻辑 def save_trajectory(self, prob, value, reward): self.saved_action_probs.append(prob) self.saved_values.append(value) self.rewards.append(reward) def clear_saved(self): self.rewards.clear() self.saved_action_probs.clear() self.saved_values.clear()
修正PPOAgent的play_game和update_policy方法
改为收集批量轨迹后再更新策略:
import torch.nn.functional as F class PPOAgent: # ...原有__init__逻辑 def play_game(self, num_episodes): update_interval = 10 # 改为每10步或 episode结束后更新 for episode in range(num_episodes): state, _, _ = self.game.reset() total_reward = 0 done = False while not done: state_tensor = torch.FloatTensor(state).unsqueeze(0).to(self.device) action_probs, value = self.policy(state_tensor) dist = Categorical(action_probs) action = dist.sample().item() next_state, reward, done = self.game.step(action) total_reward += reward # 存储轨迹数据 self.policy.save_trajectory( dist.log_prob(torch.tensor(action).to(self.device)), value, reward ) # 达到更新间隔或episode结束时更新策略 if len(self.policy.rewards) >= update_interval or done: self.update_policy() state = next_state print(f"Episode {episode+1}: Total Reward = {total_reward}") def update_policy(self): R = 0 G = [] # 计算折扣回报 for reward in reversed(self.policy.rewards): R = reward + self.gamma * R G.insert(0, R) G = torch.tensor(G, dtype=torch.float32).to(self.device) G = (G - G.mean()) / (G.std() + 1e-9) loss = 0 for log_prob, value, g in zip( self.policy.saved_action_probs, self.policy.saved_values, G ): advantage = g - value.item() # 计算actor和critic损失 actor_loss = -log_prob * advantage critic_loss = F.smooth_l1_loss(value, torch.tensor([g]).to(self.device)) loss += actor_loss + self.critic_coef * critic_loss # 添加熵正则化(可选,提升探索性) # 熵计算可以在收集轨迹时存储,这里简化处理 # entropy = dist.entropy().mean() # loss -= self.entropy_coef * entropy self.optimizer.zero_grad() loss.backward() self.optimizer.step() self.policy.clear_saved()
4. 统一action_dim参数
要么将Game类中getGameState的action_dim改为7,要么将main中的action_dim改回3,确保智能体输出的action数量与游戏支持的操作一致。
内容的提问来源于stack exchange,提问作者Keanu
相关产品推荐
相关产品推荐

