在R中基于两列唯一值动态拆分大型DataFrame为命名子DataFrame
按SalesType与LeadSale的唯一组合拆分DataFrame为多个子DataFrame
问题描述
我需要将一个DataFrame按照SalesType和LeadSale两列的唯一值组合进行拆分,生成多个小型DataFrame,每个子DataFrame仅包含对应组合的行数据。
输入数据
structure(list(BusinessDate = structure(c(1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800), class = c("POSIXct", "POSIXt"), tzone = ""), ReportType = c("Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales"), SalesType = c("Retail", "Retail", "Online", "Retail", "Online", "Online", "Market", "Market", "Retail", "Online", "Market", "Online", "Retail", "Retail", "Retail", "Retail", "Market", "Market", "Online", "Online"), Currency = c("USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD", "USD"), LeadSale = c("Mark", "Mark", "Mark", "Mark", "Jessica", "Jessica", "Jessica", "Mark", "Jessica", "Jessica", "Jessica", "Jessica", "Mark", "Mark", "Mark", "Mark", "Mark", "Mark", "Mark", "Mark"), Value = c(19, 189, 0, 0, 236, 36, 81, 19, 34, 12, 12, 12, 45.5, 45.5, 45.5, 45.5, 45.5, 45.5, 45.5, 0)), row.names = c(NA, 20L), class = "data.frame")
预期输出
生成以output_<SalesType>_<LeadSale>命名的子DataFrame,示例如下:
output_market_jessica <- structure(list(BusinessDate = structure(c(1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800), tzone = "", class = c("POSIXct", "POSIXt")), ReportType = c("Sales", "Sales", "Sales", "Sales", "Sales", "Sales"), SalesType = c("Market", "Market", "Market", "Market", "Market", "Market"), Currency = c("USD", "USD", "USD", "USD", "USD", "USD"), LeadSale = c("Jessica", "Jessica", "Jessica", "Jessica", "Jessica", "Jessica"), Value = c(236, 36, 81, 12, 12, 12)), row.names = c(NA, -6L), class = "data.frame") output_market_mark <- structure(list(BusinessDate = structure(c(1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800), tzone = "", class = c("POSIXct", "POSIXt")), ReportType = c("Sales", "Sales", "Sales", "Sales", "Sales", "Sales"), SalesType = c("Market", "Market", "Market", "Market", "Market", "Market"), Currency = c("USD", "USD", "USD", "USD", "USD", "USD"), LeadSale = c("Mark", "Mark", "Mark", "Mark", "Mark", "Mark"), Value = c(0, 19, 45.5, 45.5, 45.5, 0)), row.names = c(NA, -6L), class = "data.frame") output_retail_jessica <- structure(list(BusinessDate = structure(1672318800, tzone = "", class = c("POSIXct", "POSIXt")), ReportType = "Sales", SalesType = "Retail", Currency = "USD", LeadSale = "Jessica", Value = 34), row.names = c(NA, -1L), class = "data.frame") output_retail_mark <- structure(list(BusinessDate = structure(c(1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800, 1672318800), tzone = "", class = c("POSIXct", "POSIXt")), ReportType = c("Sales", "Sales", "Sales", "Sales", "Sales", "Sales", "Sales"), SalesType = c("Retail", "Retail", "Retail", "Retail", "Retail", "Retail", "Retail"), Currency = c("USD", "USD", "USD", "USD", "USD", "USD", "USD"), LeadSale = c("Mark", "Mark", "Mark", "Mark", "Mark", "Mark", "Mark"), Value = c(19, 189, 0, 45.5, 45.5, 45.5, 45.5)), row.names = c(NA, -7L), class = "data.frame")
解决方案
方法1:基础R实现
使用split()函数按指定列拆分DataFrame,再通过循环将每个子DataFrame赋值到全局环境,匹配预期命名规则:
# 读取输入数据 df <- structure(...) # 替换为你的输入DataFrame # 按SalesType和LeadSale组合拆分 split_list <- split(df, list(df$SalesType, df$LeadSale)) # 遍历拆分后的列表,赋值为指定名称的DataFrame for (name in names(split_list)) { # 处理名称:将"SalesType.LeadSale"转为"output_salestype_leadsale"格式 new_name <- paste0("output_", tolower(gsub("\\.", "_", name))) # 赋值到全局环境 assign(new_name, split_list[[name]], envir = .GlobalEnv) }
方法2:tidyverse工具链实现
如果习惯使用dplyr和purrr,可以用group_split()拆分,再通过pwalk()完成命名赋值:
library(dplyr) library(purrr) # 读取输入数据 df <- structure(...) # 替换为你的输入DataFrame # 获取唯一组合并生成对应名称 groups <- df %>% select(SalesType, LeadSale) %>% distinct() %>% mutate(name = paste0("output_", tolower(SalesType), "_", tolower(LeadSale))) # 拆分并赋值 df %>% group_split(SalesType, LeadSale, .keep = TRUE) %>% pwalk(function(x, nm) assign(nm, x, envir = .GlobalEnv), nm = groups$name)
两种方法都能生成符合预期的命名子DataFrame,每个子DataFrame仅包含对应SalesType和LeadSale组合的行数据。
内容的提问来源于stack exchange,提问作者alice_hooper
相关产品推荐
相关产品推荐

