You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在R中批量处理目录下所有.txt文件并合并数据框

问题:批量处理并合并多个TXT文件的数据框

目录中有多个格式一致但数据不同的.txt文件(示例为3个:sample1.txt、sample2.txt、sample3.txt),目前已能通过代码逐个处理文件生成数据框,但手动合并效率极低,需要实现自动逐个处理每个文件生成数据框,并将后续文件的数据框逐行追加到初始数据框中。

示例效果:
处理sample1.txt后的数据框:

ID picture color
1     1     red
1     2     red
1     3     blue

合并所有文件后期望得到:

ID picture color
    1     1     red
    1     2     red
    1     3     blue
    2     1     red
    2     3     blue
    2     4     green

当前处理单个文件的R代码如下:

#set up working directory, files, and save path and name
list.of.packages = c("dplyr", "berryFunctions", "stringr","gsubfn")
new.packages = list.of.packages[!(list.of.packages %in% installed.packages()[,"Package"])]
if(length(new.packages)) install.packages(new.packages)

library ("berryFunctions")
library("stringr")
library("dplyr")
library("gsubfn")
wd <- "C:/Users/PC/Desktop/New folder/GoD/" #change this to where the raw text files are located
setwd(wd)
Export_File_Name <- "GameOfDice_CATCH-018-1.txt" #name of raw file
csvname <- "output.csv" #name of save output file
files <- list.files(pattern = "*.txt") #name of files
#############################################################
#change files to utf8 and the change directory
if (file.exists("./utf8dir")) {unlink("./utf8dir", recursive = TRUE)}

convert_file_to_utf8 <- function(in_file, out_file, encoding = "utf-16") {
  in_file_conn <- file(in_file, encoding = encoding)
  txt <- readLines(in_file_conn)
  close(in_file_conn)
  
  # Create out directory
  if (!dir.exists(dirname(out_file))) dir.create(dirname(out_file))
  
  # Write file with new encoding
  out_file_conn <- file(out_file, encoding = "utf-8")
  writeLines(txt, out_file_conn)
  close(out_file_conn)
}

create_utf8_dir <- function(in_dir = "./utf16dir/", out_dir = "./utf8dir/") {
  files <- dir(in_dir, full.names = TRUE)
  for (in_file in files) {
    out_file <- sub(in_dir, out_dir, in_file, fixed = TRUE)
    convert_file_to_utf8(in_file, out_file)
  }
}
create_utf8_dir(wd)
wd <- "C:/Users/PC/Desktop/New folder/GoD/utf8dir/"
setwd(wd)
savepath <- "C:/Users/PC/Desktop/New folder/GoD/utf8dir/save/" #set to same as "wd" but with "save" added. we'll make that folder later
if (file.exists("./save")) {unlink("./save", recursive = TRUE)}
dir.create(paste(wd,"save",sep="/"))
#############################################################
# file_func <- function(wd)({
  files <- list.files(pattern = "*.txt")
  df_delim <- read.delim(files)
  df_delim <- as.data.frame(df_delim)
  colnames(df_delim) <- c("Header_Start")
  df_delim$Header_Start <- as.character(df_delim$Header_Start)
  df_delim <- subset(df_delim,Header_Start != "")
  dat <- df_delim
  
  dat <- insertRows(dat, 1 , new = NA)
  dat[1,] <- colnames(dat)
  colnames(dat) <- c("1")
  df <- dat
  df$`1` <- gsub("X....","",df$`1`)
  df$`1` <- gsub("[....^]","",df$`1`)
  df$`1` <- gsub("[***]","",df$`1`)
  df$'2' <- NA 
  df[c('1', '2')] <- str_split_fixed(df$'1', ':', 2)
  df$`1` <- gsub("X....","",df$`1`)
  df$`2` <- gsub("X....","",df$`2`)
  df$`1` <- gsub("[....^]","",df$`1`)
  df$`2` <- gsub("[....^]","",df$`2`)
  df$`1` <- gsub("[***]","",df$`1`)
  df$`2` <- gsub("[***]","",df$`2`)
  df$`1` <- gsub("\\s","",df$`1`)
  df$`2` <- gsub("\\s","",df$`2`)
  df$`1` <- gsub("\\.","_",df$`1`)
  df$`1` <- gsub("\\s","_",df$`1`)
  df$`1` <- gsub("-","_",df$`1`)
  df$`2` <- gsub("\\{","",df$`2`)
  df$`2` <- gsub("\\}","",df$`2`)
  rownames(df) <- NULL
  firstrows <- as.data.frame(df[1:24,1])
  secondrows <- as.data.frame(df[1:24,2])
  bound_rows <- as.data.frame(cbind(firstrows,secondrows))
  colnames(bound_rows) <- c("1", "2")
  df <- df[-c(1:24),]
  df <- slice(df, 1:(n() - 22))
  
  c=1
  for (row in 1:nrow(df)){
    if (df[row,'1']=='LogFrameStart' &
        df[row+1,'1'] == 'Procedure'){
      sample_data = df[row:(row+30),]
      
      if (c==1){
        newdata = sample_data
        c=c+1
      } else { newdata = cbind(newdata,sample_data[,2])}
    }
  }
  
  data1 <- newdata
  n_col <- seq(ncol(data1[,2:ncol(data1)]))
  labels <- data1[,1]
  data1 <- data1[,-1]
  colnames(data1) <- n_col
  data1 <- cbind(labels,data1)
  data1 <- data1[-1,]
  data1 <- slice(data1, 1:(n() - 2))
  data1 <- t(data1)
  colnames(data1) <- data1[1,]
  rownames(data1) <- NULL
  data1 <- data1[-1,]
  rownames(data1) <- NULL
  rownames(data1)
  data1 <- as.data.frame(data1)
  data1$Trial <- rownames(data1)
  data1 <- data1 %>%
    select(Trial, everything())
  data1$Subject <- NA
  data1 <- data1 %>%
    select(Subject, everything())
  
  fill <- as.character(strapplyc(files, "-(.*)-", simplify = TRUE))
  data1$Subject <- fill

解决方案

核心思路是将单文件处理逻辑封装为函数,遍历所有目标文件逐个处理生成数据框,最后合并所有结果。

1. 封装单文件处理函数

将原代码中处理单个文件的逻辑提取为独立函数,避免依赖全局变量,确保每个文件的处理逻辑独立。

2. 批量处理并合并

使用循环或purrr包的高效函数批量处理所有文件,自动行绑定合并数据框。

修改后的完整代码

# 安装并加载所需包
list.of.packages = c("dplyr", "berryFunctions", "stringr","gsubfn")
new.packages = list.of.packages[!(list.of.packages %in% installed.packages()[,"Package"])]
if(length(new.packages)) install.packages(new.packages)

library(berryFunctions)
library(stringr)
library(dplyr)
library(gsubfn)

# --------------------------
# 编码转换工具函数
# --------------------------
convert_file_to_utf8 <- function(in_file, out_file, encoding = "utf-16") {
  in_file_conn <- file(in_file, encoding = encoding)
  txt <- readLines(in_file_conn)
  close(in_file_conn)
  
  if (!dir.exists(dirname(out_file))) dir.create(dirname(out_file))
  
  out_file_conn <- file(out_file, encoding = "utf-8")
  writeLines(txt, out_file_conn)
  close(out_file_conn)
}

create_utf8_dir <- function(in_dir = "./", out_dir = "./utf8dir/") {
  files <- dir(in_dir, pattern = "*.txt", full.names = TRUE)
  for (in_file in files) {
    out_file <- sub(in_dir, out_dir, in_file, fixed = TRUE)
    convert_file_to_utf8(in_file, out_file)
  }
}

# --------------------------
# 单文件处理函数
# --------------------------
process_single_file <- function(file_path) {
  df_delim <- read.delim(file_path)
  df_delim <- as.data.frame(df_delim)
  colnames(df_delim) <- c("Header_Start")
  df_delim$Header_Start <- as.character(df_delim$Header_Start)
  df_delim <- subset(df_delim, Header_Start != "")
  dat <- df_delim
  
  dat <- insertRows(dat, 1 , new = NA)
  dat[1,] <- colnames(dat)
  colnames(dat) <- c("1")
  df <- dat
  df$`1` <- gsub("X....","",df$`1`)
  df$`1` <- gsub("[....^]","",df$`1`)
  df$`1` <- gsub("[***]","",df$`1`)
  df$'2' <- NA 
  df[c('1', '2')] <- str_split_fixed(df$'1', ':', 2)
  df$`1` <- gsub("X....","",df$`1`)
  df$`2` <- gsub("X....","",df$`2`)
  df$`1` <- gsub("[....^]","",df$`1`)
  df$`2` <- gsub("[....^]","",df$`2`)
  df$`1` <- gsub("[***]","",df$`1`)
  df$`2` <- gsub("[***]","",df$`2`)
  df$`1` <- gsub("\\s","",df$`1`)
  df$`2` <- gsub("\\s","",df$`2`)
  df$`1` <- gsub("\\.","_",df$`1`)
  df$`1` <- gsub("\\s","_",df$`1`)
  df$`1` <- gsub("-","_",df$`1`)
  df$`2` <- gsub("\\{","",df$`2`)
  df$`2` <- gsub("\\}","",df$`2`)
  rownames(df) <- NULL
  firstrows <- as.data.frame(df[1:24,1])
  secondrows <- as.data.frame(df[1:24,2])
  bound_rows <- as.data.frame(cbind(firstrows,secondrows))
  colnames(bound_rows) <- c("1", "2")
  df <- df[-c(1:24),]
  df <- slice(df, 1:(n() - 22))
  
  c=1
  for (row in 1:nrow(df)){
    if (df[row,'1']=='LogFrameStart' &
        df[row+1,'1'] == 'Procedure'){
      sample_data = df[row:(row+30),]
      
      if (c==1){
        newdata = sample_data
        c=c+1
      } else { newdata = cbind(newdata,sample_data[,2])}
    }
  }
  
  data1 <- newdata
  n_col <- seq(ncol(data1[,2:ncol(data1)]))
  labels <- data1[,1]
  data1 <- data1[,-1]
  colnames(data1) <- n_col
  data1 <- cbind(labels,data1)
  data1 <- data1[-1,]
  data1 <- slice(data1, 1:(n() - 2))
  data1 <- t(data1)
  colnames(data1) <- data1[1,]
  rownames(data1) <- NULL
  data1 <- data1[-1,]
  rownames(data1) <- NULL
  data1 <- as.data.frame(data1)
  data1$Trial <- rownames(data1)
  data1 <- data1 %>%
    select(Trial, everything())
  data1$Subject <- NA
  data1 <- data1 %>%
    select(Subject, everything())
  
  # 从当前文件路径提取Subject信息
  fill <- as.character(strapplyc(file_path, "-(.*)-", simplify = TRUE))
  data1$Subject <- fill
  
  return(data1)
}

# --------------------------
# 主流程:编码转换 + 批量处理合并
# --------------------------
# 设置原始文件目录
raw_wd <- "C:/Users/PC/Desktop/New folder/GoD/"
setwd(raw_wd)

# 转换所有TXT文件为UTF-8格式
if (file.exists("./utf8dir")) {unlink("./utf8dir", recursive = TRUE)}
create_utf8_dir(in_dir = raw_wd, out_dir = "./utf8dir/")

# 切换到UTF-8文件目录
utf8_wd <- "./utf8dir/"
setwd(utf8_wd)

# 创建保存目录
savepath <- "./save/"
if (file.exists(savepath)) {unlink(savepath, recursive = TRUE)}
dir.create(savepath)

# 获取所有UTF-8格式的TXT文件
all_files <- list.files(pattern = "*.txt", full.names = TRUE)

# 批量处理并合并:方式1 - 基础循环(兼容所有R版本)
merged_df <- data.frame()
for (file in all_files) {
  temp_df <- process_single_file(file)
  merged_df <- rbind(merged_df, temp_df)
}

# 批量处理并合并:方式2 - 使用purrr包高效合并(需额外加载purrr)
# library(purrr)
# merged_df <- map_dfr(all_files, process_single_file)

# 重置行名
rownames(merged_df) <- NULL

# 保存合并后的结果
write.csv(merged_df, file.path(savepath, "merged_output.csv"), row.names = FALSE)

关键修改说明

  • 将单文件处理逻辑封装为process_single_file函数,接收文件路径参数,避免全局变量依赖
  • 调整create_utf8_dir函数,适配传入的原始目录,无需硬编码路径
  • 提供两种合并方式:基础循环(无额外包依赖)和purrr::map_dfr(更简洁高效)
  • 自动保存合并后的结果到指定目录

内容的提问来源于stack exchange,提问作者jc2525

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.22 09:17:35