如何在R中批量处理目录下所有.txt文件并合并数据框
问题:批量处理并合并多个TXT文件的数据框
目录中有多个格式一致但数据不同的.txt文件(示例为3个:sample1.txt、sample2.txt、sample3.txt),目前已能通过代码逐个处理文件生成数据框,但手动合并效率极低,需要实现自动逐个处理每个文件生成数据框,并将后续文件的数据框逐行追加到初始数据框中。
示例效果:
处理sample1.txt后的数据框:
ID picture color 1 1 red 1 2 red 1 3 blue
合并所有文件后期望得到:
ID picture color 1 1 red 1 2 red 1 3 blue 2 1 red 2 3 blue 2 4 green
当前处理单个文件的R代码如下:
#set up working directory, files, and save path and name list.of.packages = c("dplyr", "berryFunctions", "stringr","gsubfn") new.packages = list.of.packages[!(list.of.packages %in% installed.packages()[,"Package"])] if(length(new.packages)) install.packages(new.packages) library ("berryFunctions") library("stringr") library("dplyr") library("gsubfn") wd <- "C:/Users/PC/Desktop/New folder/GoD/" #change this to where the raw text files are located setwd(wd) Export_File_Name <- "GameOfDice_CATCH-018-1.txt" #name of raw file csvname <- "output.csv" #name of save output file files <- list.files(pattern = "*.txt") #name of files ############################################################# #change files to utf8 and the change directory if (file.exists("./utf8dir")) {unlink("./utf8dir", recursive = TRUE)} convert_file_to_utf8 <- function(in_file, out_file, encoding = "utf-16") { in_file_conn <- file(in_file, encoding = encoding) txt <- readLines(in_file_conn) close(in_file_conn) # Create out directory if (!dir.exists(dirname(out_file))) dir.create(dirname(out_file)) # Write file with new encoding out_file_conn <- file(out_file, encoding = "utf-8") writeLines(txt, out_file_conn) close(out_file_conn) } create_utf8_dir <- function(in_dir = "./utf16dir/", out_dir = "./utf8dir/") { files <- dir(in_dir, full.names = TRUE) for (in_file in files) { out_file <- sub(in_dir, out_dir, in_file, fixed = TRUE) convert_file_to_utf8(in_file, out_file) } } create_utf8_dir(wd) wd <- "C:/Users/PC/Desktop/New folder/GoD/utf8dir/" setwd(wd) savepath <- "C:/Users/PC/Desktop/New folder/GoD/utf8dir/save/" #set to same as "wd" but with "save" added. we'll make that folder later if (file.exists("./save")) {unlink("./save", recursive = TRUE)} dir.create(paste(wd,"save",sep="/")) ############################################################# # file_func <- function(wd)({ files <- list.files(pattern = "*.txt") df_delim <- read.delim(files) df_delim <- as.data.frame(df_delim) colnames(df_delim) <- c("Header_Start") df_delim$Header_Start <- as.character(df_delim$Header_Start) df_delim <- subset(df_delim,Header_Start != "") dat <- df_delim dat <- insertRows(dat, 1 , new = NA) dat[1,] <- colnames(dat) colnames(dat) <- c("1") df <- dat df$`1` <- gsub("X....","",df$`1`) df$`1` <- gsub("[....^]","",df$`1`) df$`1` <- gsub("[***]","",df$`1`) df$'2' <- NA df[c('1', '2')] <- str_split_fixed(df$'1', ':', 2) df$`1` <- gsub("X....","",df$`1`) df$`2` <- gsub("X....","",df$`2`) df$`1` <- gsub("[....^]","",df$`1`) df$`2` <- gsub("[....^]","",df$`2`) df$`1` <- gsub("[***]","",df$`1`) df$`2` <- gsub("[***]","",df$`2`) df$`1` <- gsub("\\s","",df$`1`) df$`2` <- gsub("\\s","",df$`2`) df$`1` <- gsub("\\.","_",df$`1`) df$`1` <- gsub("\\s","_",df$`1`) df$`1` <- gsub("-","_",df$`1`) df$`2` <- gsub("\\{","",df$`2`) df$`2` <- gsub("\\}","",df$`2`) rownames(df) <- NULL firstrows <- as.data.frame(df[1:24,1]) secondrows <- as.data.frame(df[1:24,2]) bound_rows <- as.data.frame(cbind(firstrows,secondrows)) colnames(bound_rows) <- c("1", "2") df <- df[-c(1:24),] df <- slice(df, 1:(n() - 22)) c=1 for (row in 1:nrow(df)){ if (df[row,'1']=='LogFrameStart' & df[row+1,'1'] == 'Procedure'){ sample_data = df[row:(row+30),] if (c==1){ newdata = sample_data c=c+1 } else { newdata = cbind(newdata,sample_data[,2])} } } data1 <- newdata n_col <- seq(ncol(data1[,2:ncol(data1)])) labels <- data1[,1] data1 <- data1[,-1] colnames(data1) <- n_col data1 <- cbind(labels,data1) data1 <- data1[-1,] data1 <- slice(data1, 1:(n() - 2)) data1 <- t(data1) colnames(data1) <- data1[1,] rownames(data1) <- NULL data1 <- data1[-1,] rownames(data1) <- NULL rownames(data1) data1 <- as.data.frame(data1) data1$Trial <- rownames(data1) data1 <- data1 %>% select(Trial, everything()) data1$Subject <- NA data1 <- data1 %>% select(Subject, everything()) fill <- as.character(strapplyc(files, "-(.*)-", simplify = TRUE)) data1$Subject <- fill
解决方案
核心思路是将单文件处理逻辑封装为函数,遍历所有目标文件逐个处理生成数据框,最后合并所有结果。
1. 封装单文件处理函数
将原代码中处理单个文件的逻辑提取为独立函数,避免依赖全局变量,确保每个文件的处理逻辑独立。
2. 批量处理并合并
使用循环或purrr包的高效函数批量处理所有文件,自动行绑定合并数据框。
修改后的完整代码
# 安装并加载所需包 list.of.packages = c("dplyr", "berryFunctions", "stringr","gsubfn") new.packages = list.of.packages[!(list.of.packages %in% installed.packages()[,"Package"])] if(length(new.packages)) install.packages(new.packages) library(berryFunctions) library(stringr) library(dplyr) library(gsubfn) # -------------------------- # 编码转换工具函数 # -------------------------- convert_file_to_utf8 <- function(in_file, out_file, encoding = "utf-16") { in_file_conn <- file(in_file, encoding = encoding) txt <- readLines(in_file_conn) close(in_file_conn) if (!dir.exists(dirname(out_file))) dir.create(dirname(out_file)) out_file_conn <- file(out_file, encoding = "utf-8") writeLines(txt, out_file_conn) close(out_file_conn) } create_utf8_dir <- function(in_dir = "./", out_dir = "./utf8dir/") { files <- dir(in_dir, pattern = "*.txt", full.names = TRUE) for (in_file in files) { out_file <- sub(in_dir, out_dir, in_file, fixed = TRUE) convert_file_to_utf8(in_file, out_file) } } # -------------------------- # 单文件处理函数 # -------------------------- process_single_file <- function(file_path) { df_delim <- read.delim(file_path) df_delim <- as.data.frame(df_delim) colnames(df_delim) <- c("Header_Start") df_delim$Header_Start <- as.character(df_delim$Header_Start) df_delim <- subset(df_delim, Header_Start != "") dat <- df_delim dat <- insertRows(dat, 1 , new = NA) dat[1,] <- colnames(dat) colnames(dat) <- c("1") df <- dat df$`1` <- gsub("X....","",df$`1`) df$`1` <- gsub("[....^]","",df$`1`) df$`1` <- gsub("[***]","",df$`1`) df$'2' <- NA df[c('1', '2')] <- str_split_fixed(df$'1', ':', 2) df$`1` <- gsub("X....","",df$`1`) df$`2` <- gsub("X....","",df$`2`) df$`1` <- gsub("[....^]","",df$`1`) df$`2` <- gsub("[....^]","",df$`2`) df$`1` <- gsub("[***]","",df$`1`) df$`2` <- gsub("[***]","",df$`2`) df$`1` <- gsub("\\s","",df$`1`) df$`2` <- gsub("\\s","",df$`2`) df$`1` <- gsub("\\.","_",df$`1`) df$`1` <- gsub("\\s","_",df$`1`) df$`1` <- gsub("-","_",df$`1`) df$`2` <- gsub("\\{","",df$`2`) df$`2` <- gsub("\\}","",df$`2`) rownames(df) <- NULL firstrows <- as.data.frame(df[1:24,1]) secondrows <- as.data.frame(df[1:24,2]) bound_rows <- as.data.frame(cbind(firstrows,secondrows)) colnames(bound_rows) <- c("1", "2") df <- df[-c(1:24),] df <- slice(df, 1:(n() - 22)) c=1 for (row in 1:nrow(df)){ if (df[row,'1']=='LogFrameStart' & df[row+1,'1'] == 'Procedure'){ sample_data = df[row:(row+30),] if (c==1){ newdata = sample_data c=c+1 } else { newdata = cbind(newdata,sample_data[,2])} } } data1 <- newdata n_col <- seq(ncol(data1[,2:ncol(data1)])) labels <- data1[,1] data1 <- data1[,-1] colnames(data1) <- n_col data1 <- cbind(labels,data1) data1 <- data1[-1,] data1 <- slice(data1, 1:(n() - 2)) data1 <- t(data1) colnames(data1) <- data1[1,] rownames(data1) <- NULL data1 <- data1[-1,] rownames(data1) <- NULL data1 <- as.data.frame(data1) data1$Trial <- rownames(data1) data1 <- data1 %>% select(Trial, everything()) data1$Subject <- NA data1 <- data1 %>% select(Subject, everything()) # 从当前文件路径提取Subject信息 fill <- as.character(strapplyc(file_path, "-(.*)-", simplify = TRUE)) data1$Subject <- fill return(data1) } # -------------------------- # 主流程:编码转换 + 批量处理合并 # -------------------------- # 设置原始文件目录 raw_wd <- "C:/Users/PC/Desktop/New folder/GoD/" setwd(raw_wd) # 转换所有TXT文件为UTF-8格式 if (file.exists("./utf8dir")) {unlink("./utf8dir", recursive = TRUE)} create_utf8_dir(in_dir = raw_wd, out_dir = "./utf8dir/") # 切换到UTF-8文件目录 utf8_wd <- "./utf8dir/" setwd(utf8_wd) # 创建保存目录 savepath <- "./save/" if (file.exists(savepath)) {unlink(savepath, recursive = TRUE)} dir.create(savepath) # 获取所有UTF-8格式的TXT文件 all_files <- list.files(pattern = "*.txt", full.names = TRUE) # 批量处理并合并:方式1 - 基础循环(兼容所有R版本) merged_df <- data.frame() for (file in all_files) { temp_df <- process_single_file(file) merged_df <- rbind(merged_df, temp_df) } # 批量处理并合并:方式2 - 使用purrr包高效合并(需额外加载purrr) # library(purrr) # merged_df <- map_dfr(all_files, process_single_file) # 重置行名 rownames(merged_df) <- NULL # 保存合并后的结果 write.csv(merged_df, file.path(savepath, "merged_output.csv"), row.names = FALSE)
关键修改说明
- 将单文件处理逻辑封装为
process_single_file函数,接收文件路径参数,避免全局变量依赖 - 调整
create_utf8_dir函数,适配传入的原始目录,无需硬编码路径 - 提供两种合并方式:基础循环(无额外包依赖)和
purrr::map_dfr(更简洁高效) - 自动保存合并后的结果到指定目录
内容的提问来源于stack exchange,提问作者jc2525
相关产品推荐
相关产品推荐

