You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在R中实现Howard MDP理论下双策略的Sequential Value Iteration?

双策略序贯值迭代的R实现问题

我正在阅读Ronald Howard所著的《Dynamic Programming & MDP》,书中第29页的玩具制造商示例包含两个策略,每个策略对应转移概率矩阵和奖励矩阵:

# 策略1的初始变量
P1 <- matrix(c(0.5, 0.5, 0.4, 0.6), nrow = 2, byrow = TRUE);rownames(P1)=c("1","2");P1
R1 <- matrix(c(9, 3, 3, -7), nrow = 2, byrow = TRUE);rownames(R1)=c("1","2");R1

# 策略2的初始变量
P2 <- matrix(c(0.8, 0.2, 0.7, 0.3), nrow = 2, byrow = TRUE);rownames(P2)=c("1","2");P2
R2 <- matrix(c(4, 4, 1, -19), nrow = 2, byrow = TRUE);rownames(R2)=c("1","2");R2

我已经将这4个矩阵按策略合并为如下数据框:

library(dplyr)

P = as_tibble(rbind(P1,P2))%>%
  mutate(policy = rep(c(1,2),2))%>%
  arrange(policy)%>%
  rename(p1 =V1,p2 = V2);P
  
R = as_tibble(rbind(R1,R2))%>%
  mutate(policy2 = rep(c(1,2),2))%>%
  arrange(policy2)%>%
  rename(r1 =V1,r2 = V2);R
cbind(P,R)%>%
  select(-policy2)%>%
  relocate(.,policy,.after = r2)%>%
  group_by(policy)

合并结果:

# A tibble: 4 × 5
# Groups:   policy [2]
     p1    p2    r1    r2 policy
  <dbl> <dbl> <dbl> <dbl>  <dbl>
1   0.5   0.5     9     3      1
2   0.8   0.2     4     4      1
3   0.4   0.6     3    -7      2
4   0.7   0.3     1   -19      2

公式 $q_{I}^k = \sum_{j=1}{N}p_{ij}k r_{ij}^k$(k=1,2代表两个策略)指的是每个策略下P与R逐元素相乘后矩阵的行和。我需要基于上述数据框实现序贯值迭代(Sequential Value Iteration),计算公式3.3并生成目标表格。另外,书中第19页给出了单策略的实现代码:

# 单策略实现示例
P = matrix(c(0.5, 0.5, 0.4, 0.6), nrow = 2, byrow = TRUE)
R = matrix(c(9, 3, 3, -7), nrow = 2, byrow = TRUE)
Q = rowSums(P*R)
n = 5
v = matrix(0, nrow = 2, ncol = n+1) # 初始化v(0)为全0
v[,1] = 0 
# 计算不同n下的价值向量
for (i in 1:n) {
  v[,i+1] = Q + P %*% v[,i]
}
print(v)
# 输出结果
     [,1] [,2] [,3]  [,4]   [,5]    [,6]
[1,]    0    6  7.5  8.55  9.555 10.5555
[2,]    0   -3 -2.4 -1.44 -0.444  0.5556

请问如何在R中实现双策略的序贯值迭代?


双策略序贯值迭代的实现方案

步骤1:从合并数据框拆分策略矩阵

先从合并好的数据框中提取两个策略对应的转移概率矩阵和奖励矩阵:

library(dplyr)

# 生成完整合并数据框
combined_df <- cbind(P,R)%>%
  select(-policy2)%>%
  relocate(.,policy,.after = r2)

# 拆分策略1的P/R矩阵
policy1_df <- combined_df %>% filter(policy == 1)
P1_matrix <- as.matrix(policy1_df[,c("p1","p2")])
R1_matrix <- as.matrix(policy1_df[,c("r1","r2")])

# 拆分策略2的P/R矩阵
policy2_df <- combined_df %>% filter(policy == 2)
P2_matrix <- as.matrix(policy2_df[,c("p1","p2")])
R2_matrix <- as.matrix(policy2_df[,c("r1","r2")])

步骤2:计算每个策略的Q向量

根据公式计算两个策略的即时奖励向量Q:

Q1 <- rowSums(P1_matrix * R1_matrix)
Q2 <- rowSums(P2_matrix * R2_matrix)

步骤3:执行双策略序贯值迭代

初始化价值矩阵,通过循环完成迭代计算:

n_iter <- 5 # 迭代次数与单策略示例一致

# 初始化两个策略的价值矩阵(状态数×迭代步数+1)
v_policy1 <- matrix(0, nrow = 2, ncol = n_iter + 1)
v_policy2 <- matrix(0, nrow = 2, ncol = n_iter + 1)

# 迭代计算各步价值
for (i in 1:n_iter) {
  v_policy1[,i+1] <- Q1 + P1_matrix %*% v_policy1[,i]
  v_policy2[,i+1] <- Q2 + P2_matrix %*% v_policy2[,i]
}

步骤4:整理成目标表格

将结果合并为结构化数据框,方便查看对比:

# 整理策略1结果
policy1_results <- as.data.frame(v_policy1) %>%
  mutate(state = c("状态1", "状态2"), policy = "策略1") %>%
  relocate(state, policy) %>%
  rename_with(~paste0("v_", .-1), starts_with("V"))

# 整理策略2结果
policy2_results <- as.data.frame(v_policy2) %>%
  mutate(state = c("状态1", "状态2"), policy = "策略2") %>%
  relocate(state, policy) %>%
  rename_with(~paste0("v_", .-1), starts_with("V"))

# 合并最终结果
final_results <- rbind(policy1_results, policy2_results)
print(final_results)

示例输出:

state policy v_0 v_1  v_2   v_3    v_4     v_5
1 状态1   策略1   0   6  7.5  8.55  9.555 10.5555
2 状态2   策略1   0  -3 -2.4 -1.44 -0.444  0.5556
3 状态1   策略2   0   4  5.2  6.04  6.632  7.0976
4 状态2   策略2   0 -11 -8.6 -6.38 -4.466 -2.8798

可选:基于分组数据框的批量处理

如果想直接用合并后的分组数据框批量计算,可使用dplyr分组操作:

# 定义迭代函数
value_iteration <- function(df, n_iter) {
  P_mat <- as.matrix(df[,c("p1","p2")])
  R_mat <- as.matrix(df[,c("r1","r2")])
  Q <- rowSums(P_mat * R_mat)
  v <- matrix(0, nrow = nrow(df), ncol = n_iter + 1)
  for (i in 1:n_iter) {
    v[,i+1] <- Q + P_mat %*% v[,i]
  }
  return(as.data.frame(v))
}

# 分组应用迭代
final_results <- combined_df %>%
  group_by(policy) %>%
  group_modify(~ value_iteration(.x, n_iter = 5)) %>%
  mutate(state = rep(c("状态1", "状态2"), 2)) %>%
  relocate(state, policy) %>%
  rename_with(~paste0("v_", .-2), starts_with("V"))

print(final_results)

内容的提问来源于stack exchange,提问作者Homer Jay Simpson

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.29 19:12:10