tf_agents基于官方DQN教程训练自定义简单环境无法收敛求助
我已按照TensorFlow官方DQN教程完成智能体训练,成功解决Gym的CartPole-v0环境。由于reverb库不支持Windows系统,我仅在原有教程基础上去除了reverb依赖,其余逻辑与官方教程完全一致。
后续我尝试修改示例代码,训练智能体解决自定义的极简环境,但训练10000次迭代后仍无法收敛,该环境复杂度远低于10000次迭代的训练容量。
我尝试调整训练迭代次数、学习率、batch size、折扣因子等所有可调超参数,均未对结果产生改善。我期望智能体能够收敛到稳定获得+1奖励的策略(由于环境极简单,理想情况下仅需数百次迭代即可收敛),而非当前偶尔出现-1奖励的次优策略。
实际训练结果曲线如下:
(图注:橙色为回合步长,蓝色为平均奖励,X轴为训练迭代次数,范围0到10000)
代码
以下所有代码按顺序从上到下运行,我将其拆分为不同代码块以便于阅读调试。
导入依赖
import numpy as np import tf_agents as tfa import tensorflow as tf # for reproducability np.random.seed(100) tf.random.set_seed(100)
环境定义
该环境已通过validate_py_environment校验,不存在逻辑问题,可跳过阅读。
# 环境规则非常简单:初始位置为0 # 每一步可选择左移、不动、右移三个动作 # 连续右移3次到达位置=3则获胜,获得+1奖励 # 左移到位置=-3或者10步内未到达目标则失败,获得-1奖励 class SimpleGame(tfa.environments.py_environment.PyEnvironment): def __init__(self): # 0 - 左移 # 1 - 不动 # 2 - 右移 self._action_spec = tfa.specs.array_spec.BoundedArraySpec( shape = (), dtype = np.int32, minimum = 0, maximum = 2, name = 'action' ) self._observation_spec = tfa.specs.array_spec.BoundedArraySpec( shape = (1,), dtype = np.int32, minimum = -3, maximum = 3, name = 'observation' ) self._position = 0 self._step_counter = 0 def action_spec(self): return self._action_spec def observation_spec(self): return self._observation_spec def _observe(self): return np.array([self._position], dtype = np.int32) def _reset(self): self._position = 0 self._step_counter = 0 return tfa.trajectories.time_step.restart(self._observe()) def _step(self, action): if abs(self._position) >= 3 or self._step_counter >= 10: return self.reset() self._step_counter += 1 if action == 0: self._position -= 1 elif action == 1: pass elif action == 2: self._position += 1 else: raise ValueError('`action` should be 0 (left), 1 (do nothing) or 2 (right). You gave `%s`' % action) reward = 0 if self._position >= 3: reward = 1 elif self._position <= -3 or self._step_counter >= 10: reward = -1 if reward != 0: return tfa.trajectories.time_step.termination( self._observe(), reward ) else: # 游戏未结束 return tfa.trajectories.time_step.transition( self._observe(), reward = 0, discount = 1.0 ) # 环境校验无问题 tfa.environments.utils.validate_py_environment(SimpleGame(), episodes=10)
环境实例化
train_py_env = SimpleGame() test_py_env = SimpleGame() train_env = tfa.environments.tf_py_environment.TFPyEnvironment(train_py_env) test_env = tfa.environments.tf_py_environment.TFPyEnvironment(test_py_env)
智能体创建
q_network = tfa.networks.sequential.Sequential([ tf.keras.layers.Dense(16, activation = 'relu'), tf.keras.layers.Dense(3, activation = None) ]) agent = tfa.agents.dqn.dqn_agent.DqnAgent( train_env.time_step_spec(), train_env.action_spec(), q_network = q_network, optimizer = tf.keras.optimizers.Adam(), td_errors_loss_fn = tfa.utils.common.element_wise_squared_loss, n_step_update = 1 ) agent.initialize() agent.train = tfa.utils.common.function(agent.train)
策略评估函数
逻辑验证无误,可跳过阅读。
# 按照给定策略模拟若干回合,返回平均奖励和平均回合长度 def evaluate_policy(env, policy, episodes = 10): total_reward = 0.0 total_steps = 0 for ep in range(episodes): time_step = env.reset() # 初始奖励为0,保留该行仅为逻辑完整 total_reward += time_step.reward.numpy()[0] while not time_step.is_last(): action_step = policy.action(time_step) action_tensor = action_step.action action = action_tensor.numpy()[0] time_step = env.step(action) total_reward += time_step.reward.numpy()[0] total_steps += 1 average_reward = total_reward / episodes average_ep_length = total_steps / episodes return average_reward, average_ep_length # 训练前初始策略评估 avg_reward, avg_length = evaluate_policy(test_env, agent.policy) print("initial policy gives average reward of %.2f after an average %d steps" % (avg_reward, avg_length)) #> initial policy gives average reward of -1.00 after an average 10 steps
回放缓冲区定义
replay_buffer = tfa.replay_buffers.tf_uniform_replay_buffer.TFUniformReplayBuffer( data_spec = agent.collect_data_spec, batch_size = train_env.batch_size, max_length = 10000 ) replay_dataset = replay_buffer.as_dataset( num_parallel_calls = 3, sample_batch_size = 64, num_steps = 2 ).prefetch(3) def record_experience(buffer, time_step, action_step, next_time_step): buffer.add_batch( tfa.trajectories.trajectory.from_transition(time_step, action_step, next_time_step) ) replay_dataset_iterator = iter(replay_dataset)
训练流程
time_step = train_env.reset() episode_length_history = [] reward_history = [] for step in range(10000 + 1): # +1 是为了让10000次迭代也被计入图表 for _ in range(10): action_step = agent.collect_policy.action(time_step) action_tensor = action_step.action action = action_tensor.numpy()[0] new_time_step = train_env.step(action) reward = new_time_step.reward.numpy()[0] record_experience(replay_buffer, time_step, action_step, new_time_step) time_step = new_time_step training_experience, unused_diagnostics_info = next(replay_dataset_iterator) train_step = agent.train(training_experience) loss = train_step.loss print("step: %d, loss: %d" % (step, loss)) if step % 100 == 0: avg_reward, avg_length = evaluate_policy(test_env, agent.policy) print("average reward: %.2f average steps: %d" % (avg_reward, avg_length)) # 记录数据用于绘图 reward_history.append(avg_reward) episode_length_history.append(avg_length)
绘图代码
与问题无关,仅为完整展示。
import matplotlib.pyplot as plt fig, ax = plt.subplots() ax.set_xlabel('train iterations') fig.subplots_adjust(right=0.8) reward_ax = ax length_ax = ax.twinx() length_ax.set_frame_on(True) length_ax.patch.set_visible(False) length_ax.set_ylabel('episode length', color = "orange") length_ax.tick_params(axis = 'y', colors = "orange") reward_ax.set_ylabel('reward', color = "blue") reward_ax.tick_params(axis = 'y', colors = "blue") train_iterations = [i * 100 for i in range(len(reward_history))] reward_ax.plot(train_iterations, reward_history, color = "blue") length_ax.plot(train_iterations, episode_length_history, color = "orange") plt.show()
内容的提问来源于stack exchange,提问作者Gaberocksall
相关产品推荐
相关产品推荐

