如何在TensorFlow实现的Policy Gradient算法中更新策略?
Policy Gradient策略更新的神经网络修改方案
我用TensorFlow编写了Python版Policy Gradient算法,目前已完成基础部分,希望了解如何修改神经网络以实现策略更新。以下是当前代码:
import tensorflow as tf from random import * import sys import io seed(43) class Game: def __init__(self): # Target: 2 Player: 1 Empty: 0 self.grid = [[0, 0, 0, 0, 0], [0, 0, 0, 0, 0], [0, 0, 0, 0, 0], [0, 0, 0, 0, 0], [0, 0, 0, 0, 0]] def gen_target(self): while True: y = randint(0, len(self.grid) - 1) x = randint(0, len(self.grid[y]) - 1) self.grid[y][x] = 2 if self.grid[y][x] != 1 else 1 if self.grid[y][x] == 2: break return x, y def gen_player(self, x=None, y=None): if x and y: self.grid[y][x] = 1 else: y = randint(0, len(self.grid) - 1) x = randint(0, len(self.grid[y]) - 1) self.grid[y][x] = 1 return x, y @staticmethod def flatten(vector): return tf.reshape(tf.constant(vector, dtype=tf.float32), shape=(-1,)).numpy().tolist() def display(self): print(' ' + '―' * (len(self.grid[0]) * 3 - 2)) for row in self.grid: row_str = '│' + ' '.join(f'{val:3}' for val in row) + ' │' print(row_str) @staticmethod def create_nn(input_size, hidden_size, output_size): model = tf.keras.Sequential([ tf.keras.layers.Input(shape=(input_size,)), tf.keras.layers.Dense(hidden_size, activation='relu'), tf.keras.layers.Dense(output_size, activation='softmax') ]) return model def predict(self, model, state): stdout_temp = sys.stdout sys.stdout = io.StringIO() prediction = model.predict([self.flatten(state)]).tolist()[0] sys.stdout = stdout_temp return prediction @staticmethod def choose_action(output): # Choose action with probability return choices([0, 1, 2, 3], output)[0] @staticmethod def is_valid(grid, pos): if 0 <= pos[0] <= len(grid) - 1 and 0 <= pos[1] <= len(grid[pos[0]]) - 1: return True return False def is_finish(self): return True if 2. not in self.flatten(self.grid) else False def move(self, action, x, y): r = 0 # Reward new_x = x + (1 if action == 1 else -1 if action == 3 else 0) new_y = y + (1 if action == 2 else -1 if action == 0 else 0) # Verify if action is valid if 0 <= new_y <= len(self.grid) - 1 and 0 <= new_x <= len(self.grid[new_y]) - 1: self.grid[new_y][new_x] = 1 self.grid[y][x] = 0 r = 1 if self.is_finish() else 0 else: new_x, new_y = x, y # self.display() return new_x, new_y, r class Player: def __init__(self): self.game = Game() # Real position is (5,5) self.x = 4 self.y = 4 self.gamma = 0.99 self.lr = 0.01 self.game.gen_player(self.x, self.y) self.game.gen_target() self.model = self.game.create_nn(25, 1, 4) self.optimizer = tf.keras.optimizers.Adam(learning_rate=self.lr) self.train() def reset(self): self.x = 4 self.y = 4 self.game.__init__() self.game.gen_player(self.x, self.y) self.game.gen_target() def train(self, num_episodes=10): episodes = [] # Training loop for episode in range(num_episodes): episode = [] while not self.game.is_finish(): prediction = self.game.predict(self.model, self.game.grid) action = self.game.choose_action(prediction) self.x, self.y, r = self.game.move(action, self.x, self.y) episode.append([self.game.flatten(self.game.grid), action, r, 0]) for e, i in enumerate(episode): i[3] += i[2] for e2, j in enumerate(range(e + 1, len(episode))): i[3] += episode[j][2] * self.gamma ** (e2 + 1) episodes.append(episode) self.reset()
修改方案
1. 核心:实现策略梯度损失计算
Policy Gradient的核心逻辑是最大化带折扣回报权重的动作对数概率,实际训练时通过最小化该值的负数实现。需要用tf.GradientTape记录模型参数梯度,再用优化器更新参数。
2. 调整训练流程
收集完每个episode的轨迹数据后立即训练,无需先存储所有episode数据:
- 将episode中的状态、动作、折扣回报转换为TensorFlow张量
- 计算每个状态下所选动作的对数概率
- 计算损失:
-tf.reduce_mean(log_probs * discounted_rewards) - 计算梯度并更新模型参数
3. 优化神经网络结构
当前隐藏层仅1个神经元,表达能力不足,建议提升至64或128个神经元,帮助模型更好学习状态到动作的映射。
修改后的完整代码
import tensorflow as tf from random import * import sys import io seed(43) class Game: def __init__(self): # Target: 2 Player: 1 Empty: 0 self.grid = [[0, 0, 0, 0, 0], [0, 0, 0, 0, 0], [0, 0, 0, 0, 0], [0, 0, 0, 0, 0], [0, 0, 0, 0, 0]] def gen_target(self): while True: y = randint(0, len(self.grid) - 1) x = randint(0, len(self.grid[y]) - 1) self.grid[y][x] = 2 if self.grid[y][x] != 1 else 1 if self.grid[y][x] == 2: break return x, y def gen_player(self, x=None, y=None): if x and y: self.grid[y][x] = 1 else: y = randint(0, len(self.grid) - 1) x = randint(0, len(self.grid[y]) - 1) self.grid[y][x] = 1 return x, y @staticmethod def flatten(vector): return tf.reshape(tf.constant(vector, dtype=tf.float32), shape=(-1,)).numpy().tolist() def display(self): print(' ' + '―' * (len(self.grid[0]) * 3 - 2)) for row in self.grid: row_str = '│' + ' '.join(f'{val:3}' for val in row) + ' │' print(row_str) @staticmethod def create_nn(input_size, hidden_size, output_size): model = tf.keras.Sequential([ tf.keras.layers.Input(shape=(input_size,)), tf.keras.layers.Dense(hidden_size, activation='relu'), tf.keras.layers.Dense(output_size, activation='softmax') ]) return model def predict(self, model, state): # 简化predict方法,无需重定向stdout state_tensor = tf.convert_to_tensor([self.flatten(state)], dtype=tf.float32) return model(state_tensor).numpy()[0] @staticmethod def choose_action(output): # Choose action with probability return choices([0, 1, 2, 3], output)[0] @staticmethod def is_valid(grid, pos): if 0 <= pos[0] <= len(grid) - 1 and 0 <= pos[1] <= len(grid[pos[0]]) - 1: return True return False def is_finish(self): return True if 2. not in self.flatten(self.grid) else False def move(self, action, x, y): r = 0 # Reward new_x = x + (1 if action == 1 else -1 if action == 3 else 0) new_y = y + (1 if action == 2 else -1 if action == 0 else 0) # Verify if action is valid if 0 <= new_y <= len(self.grid) - 1 and 0 <= new_x <= len(self.grid[new_y]) - 1: self.grid[new_y][new_x] = 1 self.grid[y][x] = 0 r = 1 if self.is_finish() else 0 else: new_x, new_y = x, y # self.display() return new_x, new_y, r class Player: def __init__(self): self.game = Game() # Real position is (5,5) self.x = 4 self.y = 4 self.gamma = 0.99 self.lr = 0.01 self.game.gen_player(self.x, self.y) self.game.gen_target() # 提升隐藏层神经元数量至64 self.model = self.game.create_nn(25, 64, 4) self.optimizer = tf.keras.optimizers.Adam(learning_rate=self.lr) self.train() def reset(self): self.x = 4 self.y = 4 self.game.__init__() self.game.gen_player(self.x, self.y) self.game.gen_target() def train(self, num_episodes=100): # Training loop for episode in range(num_episodes): episode_data = [] while not self.game.is_finish(): state = self.game.flatten(self.game.grid) prediction = self.game.predict(self.model, self.game.grid) action = self.game.choose_action(prediction) self.x, self.y, r = self.game.move(action, self.x, self.y) episode_data.append([state, action, r]) # 计算折扣回报 discounted_rewards = [] running_add = 0 # 从后往前计算折扣回报,更高效 for r in reversed([d[2] for d in episode_data]): running_add = r + self.gamma * running_add discounted_rewards.insert(0, running_add) # 转换为张量 states = tf.convert_to_tensor([d[0] for d in episode_data], dtype=tf.float32) actions = tf.convert_to_tensor([d[1] for d in episode_data], dtype=tf.int32) discounted_rewards = tf.convert_to_tensor(discounted_rewards, dtype=tf.float32) # 标准化折扣回报(可选,有助于训练稳定) discounted_rewards = (discounted_rewards - tf.reduce_mean(discounted_rewards)) / (tf.math.reduce_std(discounted_rewards) + 1e-8) # 计算梯度并更新模型 with tf.GradientTape() as tape: # 得到每个状态的动作概率分布 action_probs = self.model(states) # 获取所选动作的概率 action_mask = tf.one_hot(actions, depth=4) selected_probs = tf.reduce_sum(action_probs * action_mask, axis=1) # 计算对数概率 log_probs = tf.math.log(selected_probs + 1e-8) # 加小值避免log(0) # 计算损失 loss = -tf.reduce_mean(log_probs * discounted_rewards) # 计算梯度 gradients = tape.gradient(loss, self.model.trainable_variables) # 更新参数 self.optimizer.apply_gradients(zip(gradients, self.model.trainable_variables)) print(f"Episode {episode+1}, Loss: {loss.numpy():.4f}") self.reset() if __name__ == "__main__": player = Player()
关键修改说明
- 简化
predict方法,去掉不必要的stdout重定向,直接用TensorFlow张量计算 - 调整折扣回报计算方式,从后往前累加更高效
- 加入梯度计算与模型参数更新的核心逻辑
- 提升隐藏层神经元数量,增强模型表达能力
- 可选加入折扣回报标准化,帮助训练更稳定
内容的提问来源于stack exchange,提问作者JFR001
相关产品推荐
相关产品推荐

