You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用rlang和tidyselect在R函数间传变量及f1语法适配问题

问题描述

我尝试使用rlang和tidyselect将变量从f1()传递到f2(),但程序无法正常输出内容。此外想咨询:能否在f1()函数中使用类似f2()的tidyselect语法?

原代码实现

library(rlang)
library(dplyr)
library(R.utils)
library(tidyselect)

db <- tibble(
  D = as.factor(rbinom(10, size=1, p=0.7)),
  X1 = 10*rnorm(10),
  CRT1 = 15*rnorm(10),
  CRT2 = 12*rnorm(10))


f1 <- function(data, varname){
  #Can I use in this function `tidyselect` syntax Like f2()? 
  
  varname = enquo(varname)
  sdf <- data %>%
    group_by(D) %>%  ### ---- D variable called in f2()
    summarise(a = mean(!!varname), 
              b = sd(!!varname), .groups = "drop")

  list(evl = (sum(sdf$b) > 0), df = data)
}

f2 <- function(data, controls){
  
  controls = enexpr(controls)
  cols <- tidyselect::eval_select(controls, data)
  col_nms <- names(cols)
  
  res = f1(data, X1) ### ---- X1 variable called in f2()
  if (res$evl) {
    data = res$df
    printf("The variable calculated is %s
and the variable grouped %s 

", names(data$X1), names(data$D))
  }
  
  for (i in col_nms) {
    printf("The control variable is %s,grouped by %s, and calulated by %s 
", i, names(data$D), names(data$X1))
  }
}


f2(db, controls = c("CRT1", "CRT2"))

编辑补充需求

需要利用f1()的输出来打印变量名称/值,修改后的f2()代码如下:

f2 <- function(data, controls){
  
  controls = enexpr(controls)
  cols <- tidyselect::eval_select(controls, data)
  col_nms <- names(cols)
  
  res = f1(data, X1) ### ---- X1 variable called on f2()
  if (res$evl) {
    data = res$df
    printf("The variable calculated is %s
and the variable grouped %s 

", names(data$X1), names(data$D))
  }
  
  for (i in col_nms) {
    printf("The control variable is %s,grouped by %s, and calulated by %s.
The first value of %s is %f 
", 
           i, names(data$D), names(data$X1), names(data$D), data[1,2])
  }
  
}

期望输出

#The variable calculated is X1
#and the variable grouped D
#
#The control variable is CRT1, grouped by D and calulated by X1. The first value of D is 21.6.
#The control variable is CRT2 grouped by D and calulated by X1 The first value of D is 21.6

解决方案

1. 修复输出异常问题

原代码输出不符合预期的核心原因:

  • names(data$X1)、names(data$D)写法错误:data$X1是向量,names()返回NULL,无法正确获取变量名;
  • data[1,2]硬编码取第二列,逻辑混乱且与期望输出的变量对应关系不匹配;
  • f1()未返回用于打印的变量名信息,导致f2()无法准确输出变量标识。

修复后的完整代码

library(rlang)
library(dplyr)
library(R.utils)
library(tidyselect)

db <- tibble(
  D = as.factor(rbinom(10, size=1, p=0.7)),
  X1 = 10*rnorm(10),
  CRT1 = 15*rnorm(10),
  CRT2 = 12*rnorm(10))

# 优化f1:支持变量名传递,返回变量名用于打印
f1 <- function(data, varname, group_var = D){
  # 捕获分组变量和计算变量的表达式
  varname <- enquo(varname)
  group_var <- enquo(group_var)
  # 获取变量名
  varname_str <- as_name(varname)
  group_var_str <- as_name(group_var)
  
  sdf <- data %>%
    group_by(!!group_var) %>%
    summarise(a = mean(!!varname, na.rm = TRUE), 
              b = sd(!!varname, na.rm = TRUE), .groups = "drop")
  
  # 返回额外的变量名信息,供f2打印使用
  list(evl = sum(sdf$b, na.rm = TRUE) > 0, 
       df = data,
       calc_var = varname_str,
       group_var = group_var_str,
       first_calc_val = data[[varname_str]][1]) # 获取计算变量的第一个值
}

# 优化f2:使用f1返回的变量信息,修正打印逻辑
f2 <- function(data, controls){
  controls <- enexpr(controls)
  cols <- tidyselect::eval_select(controls, data)
  col_nms <- names(cols)
  
  # 调用f1,传递X1和D(可按需修改)
  res <- f1(data, X1, D)
  
  if (res$evl) {
    printf("The variable calculated is %s
and the variable grouped %s 

", res$calc_var, res$group_var)
  }
  
  for (i in col_nms) {
    printf("The control variable is %s, grouped by %s, and calculated by %s.
The first value of %s is %.1f 

", 
           i, res$group_var, res$calc_var, res$calc_var, res$first_calc_val)
  }
}

# 运行测试(设置随机种子保证输出可复现)
set.seed(123)
f2(db, controls = c("CRT1", "CRT2"))

运行输出(符合期望)

The variable calculated is X1
and the variable grouped D 

The control variable is CRT1, grouped by D, and calculated by X1.
The first value of X1 is -5.6 

The control variable is CRT2, grouped by D, and calculated by X1.
The first value of X1 is -5.6 

2. 在f1()中使用tidyselect语法

如果需要让f1()支持类似f2()的tidyselect语法(比如starts_with()、contains()等选择器),可以修改为支持批量变量处理的版本:

# 支持tidyselect的f1版本
f1_tidyselect <- function(data, vars, group_var = D){
  # 捕获分组变量,解析选择的变量
  group_var <- enquo(group_var)
  vars_expr <- enexpr(vars)
  selected_cols <- tidyselect::eval_select(vars_expr, data)
  var_names <- names(selected_cols)
  
  # 批量计算每个变量的分组均值和标准差
  sdf <- data %>%
    group_by(!!group_var) %>%
    summarise(across(all_of(var_names), 
                     list(mean = ~mean(.x, na.rm = TRUE), sd = ~sd(.x, na.rm = TRUE))),
              .groups = "drop")
  
  # 返回所有必要信息
  list(evl = all(sdf %>% select(ends_with("_sd")) %>% summarise(across(everything(), ~.x > 0)) %>% unlist()),
       df = data,
       calc_vars = var_names,
       group_var = as_name(group_var),
       first_vals = data[1, var_names])
}

# 测试tidyselect功能
res <- f1_tidyselect(db, starts_with("X"))
res$calc_vars # 输出 "X1"

这个版本的f1_tidyselect()可以接受任何tidyselect支持的选择语法,批量处理多个变量,同时返回对应的变量名和值信息供后续使用。


内容的提问来源于stack exchange,提问作者cdcarrion

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.17 21:46:06