R中多重插补ANOVA的子集化问题技术求助
Let's work through your problem with subsetting multiply imputed data for ANOVA. Here are three practical approaches that should solve this:
1. Subset the Datlist Directly (No Need to Convert Back to MIDS)
The mi.anova function actually accepts a list of data frames (datlist) as input, so you don't need to convert your subsetted datlist back to a mids object. Here's how to do it:
# Assume your imputed data is stored as a mids object called `imputed_data` # Step 1: Convert mids to a list of completed datasets dat_list <- lapply(1:imputed_data$m, function(i) mice::complete(imputed_data, i)) # Step 2: Subset each dataset in the list (e.g., filter rows where group == "A") dat_list_subset <- lapply(dat_list, function(df) { df[df$group == "A", ] # Replace with your subset condition }) # Step 3: Run mi.anova on the subset datlist mi.anova(mi.res = dat_list_subset, formula = your_formula_here, type = 2)
This works because mi.anova checks the class of mi.res and handles datlists natively (you can see this in the original function code where it processes lists of data frames).
2. Use mice::with() + pool() (For More Control)
If you tried with() and pool() but didn't get the expected output, you might have missed how to properly subset within the with() call. Here's a step-by-step example, including handling Type 2 ANOVA:
# Fit the model on subsetted data for each imputation fit_subset <- with(imputed_data, { # Subset the current imputed dataset df_sub <- df[df$group == "A", ] # Your subset condition # Fit the linear model lm(your_formula_here, data = df_sub) }) # Pool the results pooled_fit <- pool(fit_subset) # Extract ANOVA stats from each imputation and combine anova_results <- lapply(fit_subset$result, function(mod) summary(aov(mod))[[1]]) F_vals <- sapply(anova_results, function(x) x$"F value")[1:(nrow(anova_results[[1]])-1), ] df1 <- anova_results[[1]]$Df[1:(nrow(anova_results[[1]])-1)] df2 <- anova_results[[1]]$Df[nrow(anova_results[[1]])] # Combine using miceadds::micombine.F (same logic as mi.anova) combined_stats <- t(sapply(1:length(df1), function(i) { micombine.F(F_vals[i, ], df1 = df1[i], df2 = df2, display = FALSE) })) # Build the final ANOVA table anova_table <- data.frame( df1 = df1, df2 = df2, `F value` = round(combined_stats[, 3], 4), `Pr(>F)` = round(combined_stats[, 4], 6), row.names = rownames(anova_results[[1]])[1:(nrow(anova_results[[1]])-1)] ) # Add optional SS, eta², and partial eta² SS_vals <- sapply(anova_results, function(x) x$"Sum Sq") SS_mean <- rowMeans(SS_vals) anova_table <- cbind(SSQ = SS_mean[1:(length(SS_mean)-1)], anova_table) total_SS <- sum(SS_mean) anova_table$eta2 <- round(SS_mean[1:(length(SS_mean)-1)] / total_SS, 6) anova_table$partial.eta2 <- round(SS_mean[1:(length(SS_mean)-1)] / (SS_mean[1:(length(SS_mean)-1)] + SS_mean[length(SS_mean)]), 6) print(anova_table)
For Type 3 ANOVA, modify the with() call to use car::Anova directly, then combine results using the same micombine.F approach.
3. Modified mi.anova Function with Subset Support
If you prefer a native solution, here's a modified version of mi.anova that adds a subset parameter. This applies your subset condition to each imputed dataset before running ANOVA:
mi.anova_subset <- function (mi.res, formula, type = 2, subset = NULL) { if (type == 3) { TAM::require_namespace_msg("car") } mi.list <- mi.res # Handle mids objects: complete each dataset and apply subset if (class(mi.list) == "mids.1chain") { mi.list <- mi.list$midsobj } if (class(mi.list) == "mids") { m <- mi.list$m h1 <- vector("list", m) for (ii in 1:m) { h1[[ii]] <- as.data.frame(mice::complete(mi.list, ii)) # Apply subset if provided if (!is.null(subset)) { h1[[ii]] <- subset(h1[[ii]], subset = eval(parse(text = subset))) } } mi.list <- h1 } # Handle mi.norm objects: apply subset to each imputed dataset if (class(mi.res) == "mi.norm") { mi.list <- mi.list$imp.data if (!is.null(subset)) { mi.list <- lapply(mi.list, function(df) subset(df, subset = eval(parse(text = subset)))) } } # Rest of original mi.anova code remains unchanged if (type == 2) { anova.imp0 <- lapply(mi.list, FUN = function(dat) { stats::lm(formula, data = dat) }) anova.imp <- lapply(anova.imp0, FUN = function(obj) { summary(stats::aov(obj)) }) } if (type == 3) { Nimp <- length(mi.list) vars <- all.vars(stats::as.formula(formula))[-1] VV <- length(vars) ma_contrasts <- as.list(1:VV) names(ma_contrasts) <- vars dat <- mi.list[[1]] for (vv in 1:VV) { ma_contrasts[[vars[vv]]] <- "contr.sum" if (!is.factor(dat[, vars[vv]])) { ma_contrasts[[vars[vv]]] <- NULL } } anova.imp0 <- lapply(as.list(1:Nimp), FUN = function(ii) { dat <- mi.list[[ii]] mod1 <- stats::lm(formula, data = dat, contrasts = ma_contrasts) return(mod1) }) anova.imp <- lapply(as.list(1:Nimp), FUN = function(ii) { obj <- anova.imp0[[ii]] car::Anova(obj, type = 3) }) } if (type == 2) { FF <- nrow(anova.imp[[1]][[1]]) - 1 } if (type == 3) { FF <- nrow(anova.imp[[1]]["F value"]) - 2 } anova.imp.inf <- t(sapply(1:FF, FUN = function(ff) { micombine.F(sapply(1:(length(anova.imp)), FUN = function(ii) { if (type == 2) { r1 <- anova.imp[[ii]][[1]]$"F value"[ff] } if (type == 3) { r1 <- anova.imp[[ii]]$"F value"[ff + 1] } return(r1) }), df1 = ifelse(type == 2, anova.imp[[1]][[1]]$Df[ff], anova.imp[[1]]["Df"][ff + 1, 1]), display = FALSE) })) res <- anova.imp.inf[, c(3, 4, 1, 2)] res <- matrix(res, ncol = 4) res[, 3] <- round(res[, 3], 4) res[, 4] <- round(res[, 4], 6) g1 <- rownames(anova.imp[[1]][[1]])[1:FF] if (type == 3) { g1 <- rownames(anova.imp[[1]])[1 + 1:FF] } rownames(res) <- g1 res <- data.frame(res) if (type == 2) { SS <- rowMeans(matrix(unlist(lapply(anova.imp, FUN = function(ll) { ll[[1]][, 2] })), ncol = length(mi.list))) } if (type == 3) { SS <- rowMeans(matrix(unlist(lapply(anova.imp, FUN = function(ll) { l2 <- ll["Sum Sq"][-1, 1] return(l2) })), ncol = length(mi.list))) } r.squared <- sum(SS[-(FF + 1)])/sum(SS) res$eta2 <- round(SS[-(FF + 1)]/sum(SS), 6) res$partial.eta2 <- round(SS[-(FF + 1)]/(SS[-(FF + 1)] + SS[FF + 1]), 6) g1 <- c("F value", "Pr(>F)") colnames(res)[3:4] <- g1 colnames(res)[1:2] <- c("df1", "df2") c1 <- colnames(res) res <- rbind(res, res[1, ]) rownames(res)[nrow(res)] <- "Residual" res[nrow(res), ] <- NA res <- data.frame(SSQ = SS, res) colnames(res)[-1] <- c1 # Print modified output with subset info cat("Univariate ANOVA for Multiply Imputed Data (Subset)", paste0("(Type ", type, ")"), " \n\n") cat("lm Formula: ", formula) if (!is.null(subset)) cat("\nSubset Condition: ", subset) cat(paste("\nR^2=", round(r.squared, 4), sep = ""), "\n") cat("..........................................................................\n") cat("ANOVA Table \n") print(round(res, 5)) invisible(list(r.squared = r.squared, anova.table = res, type = type)) }
How to Use the Modified Function
Call it with your subset condition as a string:
mi.anova_subset( mi.res = imputed_data, formula = y ~ x1 + x2, type = 2, subset = "x1 == 'Control' & x2 > 5" # Your subset condition )
内容的提问来源于stack exchange,提问作者P.Stegmann

