You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在TensorFlow实现的Policy Gradient算法中更新策略?

Policy Gradient策略更新的神经网络修改方案

我用TensorFlow编写了Python版Policy Gradient算法,目前已完成基础部分,希望了解如何修改神经网络以实现策略更新。以下是当前代码:

import tensorflow as tf
from random import *
import sys
import io

seed(43)


class Game:
    def __init__(self):
        # Target: 2    Player: 1    Empty: 0
        self.grid = [[0, 0, 0, 0, 0],
                     [0, 0, 0, 0, 0],
                     [0, 0, 0, 0, 0],
                     [0, 0, 0, 0, 0],
                     [0, 0, 0, 0, 0]]

    def gen_target(self):
        while True:
            y = randint(0, len(self.grid) - 1)
            x = randint(0, len(self.grid[y]) - 1)
            self.grid[y][x] = 2 if self.grid[y][x] != 1 else 1
            if self.grid[y][x] == 2:
                break
        return x, y

    def gen_player(self, x=None, y=None):
        if x and y:
            self.grid[y][x] = 1
        else:
            y = randint(0, len(self.grid) - 1)
            x = randint(0, len(self.grid[y]) - 1)
            self.grid[y][x] = 1
            return x, y

    @staticmethod
    def flatten(vector):
        return tf.reshape(tf.constant(vector, dtype=tf.float32), shape=(-1,)).numpy().tolist()

    def display(self):
        print(' ' + '―' * (len(self.grid[0]) * 3 - 2))
        for row in self.grid:
            row_str = '│' + ' '.join(f'{val:3}' for val in row) + '  │'
            print(row_str)

    @staticmethod
    def create_nn(input_size, hidden_size, output_size):
        model = tf.keras.Sequential([
            tf.keras.layers.Input(shape=(input_size,)),
            tf.keras.layers.Dense(hidden_size, activation='relu'),
            tf.keras.layers.Dense(output_size, activation='softmax')
        ])
        return model

    def predict(self, model, state):
        stdout_temp = sys.stdout
        sys.stdout = io.StringIO()
        prediction = model.predict([self.flatten(state)]).tolist()[0]
        sys.stdout = stdout_temp
        return prediction

    @staticmethod
    def choose_action(output):
        # Choose action with probability
        return choices([0, 1, 2, 3], output)[0]

    @staticmethod
    def is_valid(grid, pos):
        if 0 <= pos[0] <= len(grid) - 1 and 0 <= pos[1] <= len(grid[pos[0]]) - 1:
            return True
        return False

    def is_finish(self):
        return True if 2. not in self.flatten(self.grid) else False

    def move(self, action, x, y):
        r = 0  # Reward
        new_x = x + (1 if action == 1 else -1 if action == 3 else 0)
        new_y = y + (1 if action == 2 else -1 if action == 0 else 0)

        # Verify if action is valid
        if 0 <= new_y <= len(self.grid) - 1 and 0 <= new_x <= len(self.grid[new_y]) - 1:
            self.grid[new_y][new_x] = 1
            self.grid[y][x] = 0
            r = 1 if self.is_finish() else 0
        else:
            new_x, new_y = x, y
        # self.display()
        return new_x, new_y, r


class Player:
    def __init__(self):
        self.game = Game()
        # Real position is (5,5)
        self.x = 4
        self.y = 4
        self.gamma = 0.99
        self.lr = 0.01
        self.game.gen_player(self.x, self.y)
        self.game.gen_target()
        self.model = self.game.create_nn(25, 1, 4)
        self.optimizer = tf.keras.optimizers.Adam(learning_rate=self.lr)
        self.train()

    def reset(self):
        self.x = 4
        self.y = 4
        self.game.__init__()
        self.game.gen_player(self.x, self.y)
        self.game.gen_target()

    def train(self, num_episodes=10):
        episodes = []
        # Training loop
        for episode in range(num_episodes):
            episode = []
            while not self.game.is_finish():
                prediction = self.game.predict(self.model, self.game.grid)
                action = self.game.choose_action(prediction)
                self.x, self.y, r = self.game.move(action, self.x, self.y)
                episode.append([self.game.flatten(self.game.grid), action, r, 0])
            for e, i in enumerate(episode):
                i[3] += i[2]
                for e2, j in enumerate(range(e + 1, len(episode))):
                    i[3] += episode[j][2] * self.gamma ** (e2 + 1)
            episodes.append(episode)
            self.reset()

修改方案

1. 核心:实现策略梯度损失计算

Policy Gradient的核心逻辑是最大化带折扣回报权重的动作对数概率,实际训练时通过最小化该值的负数实现。需要用tf.GradientTape记录模型参数梯度,再用优化器更新参数。

2. 调整训练流程

收集完每个episode的轨迹数据后立即训练,无需先存储所有episode数据:

  • 将episode中的状态、动作、折扣回报转换为TensorFlow张量
  • 计算每个状态下所选动作的对数概率
  • 计算损失:-tf.reduce_mean(log_probs * discounted_rewards)
  • 计算梯度并更新模型参数

3. 优化神经网络结构

当前隐藏层仅1个神经元,表达能力不足,建议提升至64或128个神经元,帮助模型更好学习状态到动作的映射。

修改后的完整代码

import tensorflow as tf
from random import *
import sys
import io

seed(43)


class Game:
    def __init__(self):
        # Target: 2    Player: 1    Empty: 0
        self.grid = [[0, 0, 0, 0, 0],
                     [0, 0, 0, 0, 0],
                     [0, 0, 0, 0, 0],
                     [0, 0, 0, 0, 0],
                     [0, 0, 0, 0, 0]]

    def gen_target(self):
        while True:
            y = randint(0, len(self.grid) - 1)
            x = randint(0, len(self.grid[y]) - 1)
            self.grid[y][x] = 2 if self.grid[y][x] != 1 else 1
            if self.grid[y][x] == 2:
                break
        return x, y

    def gen_player(self, x=None, y=None):
        if x and y:
            self.grid[y][x] = 1
        else:
            y = randint(0, len(self.grid) - 1)
            x = randint(0, len(self.grid[y]) - 1)
            self.grid[y][x] = 1
            return x, y

    @staticmethod
    def flatten(vector):
        return tf.reshape(tf.constant(vector, dtype=tf.float32), shape=(-1,)).numpy().tolist()

    def display(self):
        print(' ' + '―' * (len(self.grid[0]) * 3 - 2))
        for row in self.grid:
            row_str = '│' + ' '.join(f'{val:3}' for val in row) + '  │'
            print(row_str)

    @staticmethod
    def create_nn(input_size, hidden_size, output_size):
        model = tf.keras.Sequential([
            tf.keras.layers.Input(shape=(input_size,)),
            tf.keras.layers.Dense(hidden_size, activation='relu'),
            tf.keras.layers.Dense(output_size, activation='softmax')
        ])
        return model

    def predict(self, model, state):
        # 简化predict方法,无需重定向stdout
        state_tensor = tf.convert_to_tensor([self.flatten(state)], dtype=tf.float32)
        return model(state_tensor).numpy()[0]

    @staticmethod
    def choose_action(output):
        # Choose action with probability
        return choices([0, 1, 2, 3], output)[0]

    @staticmethod
    def is_valid(grid, pos):
        if 0 <= pos[0] <= len(grid) - 1 and 0 <= pos[1] <= len(grid[pos[0]]) - 1:
            return True
        return False

    def is_finish(self):
        return True if 2. not in self.flatten(self.grid) else False

    def move(self, action, x, y):
        r = 0  # Reward
        new_x = x + (1 if action == 1 else -1 if action == 3 else 0)
        new_y = y + (1 if action == 2 else -1 if action == 0 else 0)

        # Verify if action is valid
        if 0 <= new_y <= len(self.grid) - 1 and 0 <= new_x <= len(self.grid[new_y]) - 1:
            self.grid[new_y][new_x] = 1
            self.grid[y][x] = 0
            r = 1 if self.is_finish() else 0
        else:
            new_x, new_y = x, y
        # self.display()
        return new_x, new_y, r


class Player:
    def __init__(self):
        self.game = Game()
        # Real position is (5,5)
        self.x = 4
        self.y = 4
        self.gamma = 0.99
        self.lr = 0.01
        self.game.gen_player(self.x, self.y)
        self.game.gen_target()
        # 提升隐藏层神经元数量至64
        self.model = self.game.create_nn(25, 64, 4)
        self.optimizer = tf.keras.optimizers.Adam(learning_rate=self.lr)
        self.train()

    def reset(self):
        self.x = 4
        self.y = 4
        self.game.__init__()
        self.game.gen_player(self.x, self.y)
        self.game.gen_target()

    def train(self, num_episodes=100):
        # Training loop
        for episode in range(num_episodes):
            episode_data = []
            while not self.game.is_finish():
                state = self.game.flatten(self.game.grid)
                prediction = self.game.predict(self.model, self.game.grid)
                action = self.game.choose_action(prediction)
                self.x, self.y, r = self.game.move(action, self.x, self.y)
                episode_data.append([state, action, r])
            
            # 计算折扣回报
            discounted_rewards = []
            running_add = 0
            # 从后往前计算折扣回报,更高效
            for r in reversed([d[2] for d in episode_data]):
                running_add = r + self.gamma * running_add
                discounted_rewards.insert(0, running_add)
            
            # 转换为张量
            states = tf.convert_to_tensor([d[0] for d in episode_data], dtype=tf.float32)
            actions = tf.convert_to_tensor([d[1] for d in episode_data], dtype=tf.int32)
            discounted_rewards = tf.convert_to_tensor(discounted_rewards, dtype=tf.float32)
            
            # 标准化折扣回报(可选,有助于训练稳定)
            discounted_rewards = (discounted_rewards - tf.reduce_mean(discounted_rewards)) / (tf.math.reduce_std(discounted_rewards) + 1e-8)
            
            # 计算梯度并更新模型
            with tf.GradientTape() as tape:
                # 得到每个状态的动作概率分布
                action_probs = self.model(states)
                # 获取所选动作的概率
                action_mask = tf.one_hot(actions, depth=4)
                selected_probs = tf.reduce_sum(action_probs * action_mask, axis=1)
                # 计算对数概率
                log_probs = tf.math.log(selected_probs + 1e-8)  # 加小值避免log(0)
                # 计算损失
                loss = -tf.reduce_mean(log_probs * discounted_rewards)
            
            # 计算梯度
            gradients = tape.gradient(loss, self.model.trainable_variables)
            # 更新参数
            self.optimizer.apply_gradients(zip(gradients, self.model.trainable_variables))
            
            print(f"Episode {episode+1}, Loss: {loss.numpy():.4f}")
            self.reset()

if __name__ == "__main__":
    player = Player()

关键修改说明

  • 简化predict方法,去掉不必要的stdout重定向,直接用TensorFlow张量计算
  • 调整折扣回报计算方式,从后往前累加更高效
  • 加入梯度计算与模型参数更新的核心逻辑
  • 提升隐藏层神经元数量,增强模型表达能力
  • 可选加入折扣回报标准化,帮助训练更稳定

内容的提问来源于stack exchange,提问作者JFR001

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.11 20:41:00