在R语言中根据a_or_b列值筛选对应abc列值生成新数据框
Let's walk through the steps to finish creating df2 using your existing index matrix approach:
Step 1: Isolate ABC-related columns
First, we'll extract only the columns starting with abc_ since we don't need the def_ columns for this task:
abc_cols <- df[, grep("^abc_", colnames(df))]
Step 2: Generate row-specific indices
Your sapply code gives us a list where each element is the set of column indices (for ah or bh) corresponding to each row's a_or_b value. We'll convert this list to a matrix for easier indexing:
row_index_list <- sapply(paste0(df$a_or_b, "h"), function(pattern) { grep(pattern, colnames(abc_cols)) }) index_matrix <- do.call(rbind, row_index_list)
Now index_matrix is a 5x3 matrix where each row holds the positions of the correct ah/bh columns for the three groups (one, two, three).
Step 3: Extract the combo values
We'll use the index matrix to pull the right values from abc_cols for each row and group:
combo_data <- sapply(1:ncol(index_matrix), function(col) { abc_cols[cbind(1:nrow(df), index_matrix[, col])] }) colnames(combo_data) <- paste0("combo_", c("one", "two", "three"))
Step 4: Assemble the final data frame
Combine the core columns (name, a_or_b) with our new combo columns:
df2 <- data.frame(df[, c("name", "a_or_b")], combo_data)
Full Base R Code
# Original data frame setup name <- c("Fred","Mark","Jen","Simon","Ed") a_or_b <- c("a","a","b","a","b") abc_ah_one <- c(3,5,2,4,7) abc_bh_one <- c(5,4,1,9,8) abc_ah_two <- c(2,1,3,7,6) abc_bh_two <- c(3,6,8,8,5) abc_ah_three <- c(5,4,7,6,2) abc_bh_three <- c(9,7,2,1,4) def_ah_one <- c(1,3,9,2,7) def_bh_one <- c(2,8,4,6,1) def_ah_two <- c(4,7,3,2,5) def_bh_two <- c(5,2,9,8,3) def_ah_three <- c(8,5,3,5,2) def_bh_three <- c(2,7,4,3,0) df <- data.frame(name,a_or_b,abc_ah_one,abc_bh_one,abc_ah_two,abc_bh_two, abc_ah_three,abc_bh_three,def_ah_one,def_bh_one, def_ah_two,def_bh_two,def_ah_three,def_bh_three) # Create df2 abc_cols <- df[, grep("^abc_", colnames(df))] row_index_list <- sapply(paste0(df$a_or_b, "h"), function(pattern) { grep(pattern, colnames(abc_cols)) }) index_matrix <- do.call(rbind, row_index_list) combo_data <- sapply(1:ncol(index_matrix), function(col) { abc_cols[cbind(1:nrow(df), index_matrix[, col])] }) colnames(combo_data) <- paste0("combo_", c("one", "two", "three")) df2 <- data.frame(df[, c("name", "a_or_b")], combo_data)
If you prefer a more readable reshaping workflow using dplyr and tidyr:
library(dplyr) library(tidyr) df2 <- df %>% select(name, a_or_b, starts_with("abc_")) %>% pivot_longer(cols = starts_with("abc_"), names_to = c(".value", "group"), names_pattern = "abc_(.*)_(.*)") %>% mutate(combo = ifelse(a_or_b == "a", ah, bh)) %>% select(-ah, -bh) %>% pivot_wider(names_from = group, values_from = combo, names_prefix = "combo_")
Both methods will produce your desired df2:
name a_or_b combo_one combo_two combo_three 1 Fred a 3 2 5 2 Mark a 5 1 4 3 Jen b 1 8 2 4 Simon a 4 7 6 5 Ed b 8 5 4
内容的提问来源于stack exchange,提问作者orangeman51

