Python多txt指定列合并为单csv/txt文件代码问题求助
问题修正方案
原代码核心问题
- 遍历scan文件夹时从索引1开始,直接漏掉了第一个符合命名规则的scan文件夹
- 合并逻辑完全错误:原代码是按行追加多个文件的内容,需求是按列拼接不同文件的指定列
- 第一版代码未保留第一个文件的前两列数据,且行循环从1开始丢失了第一行数据
- 第二版代码误用
DictWriter按行写入,完全不符合列合并的要求 - 频繁切换工作目录容易引发路径错误,更推荐用绝对路径拼接代替
os.chdir
修正后代码(原生Python版,无需额外依赖)
import os import glob # 根目录路径,替换为你自己的根目录 root_dir = "W:/certaindirectory/" # txt文件所在的子目录相对路径,替换为你scan文件夹下存放txt的实际子目录路径 txt_subpath = "xxx/xxx/" # 筛选所有scan开头的文件夹 scan_folders = [f for f in os.listdir(root_dir) if f.startswith("scan")] for folder in scan_folders: # 拼接当前scan文件夹下的txt目录绝对路径 txt_dir = os.path.join(root_dir, folder, txt_subpath) # 获取所有txt文件,排序保证列顺序固定 txt_files = sorted(glob.glob(os.path.join(txt_dir, "*.txt"))) if not txt_files: continue # 构造表头:前两列X/Y,后续列用对应txt文件名 col_names = ["X", "Y"] + [os.path.basename(f) for f in txt_files] # 输出文件路径,生成在当前scan目录下,命名为当前scan文件夹名.csv output_path = os.path.join(root_dir, folder, f"{folder}.csv") # 打开所有待读取的txt文件 file_handlers = [open(f, "r", encoding="utf-8") for f in txt_files] with open(output_path, "w", encoding="utf-8", newline="") as f_out: # 先写表头 f_out.write(",".join(col_names) + "\n") # 逐行读取拼接 for _ in range(3371): # 取第一个文件的三列作为基础 first_line = file_handlers[0].readline().strip().split("\t") row_data = first_line[:3] # 依次取剩余文件的第三列拼接 for fh in file_handlers[1:]: line = fh.readline().strip().split("\t") row_data.append(line[2]) # 写入当前行 f_out.write(",".join(row_data) + "\n") # 关闭所有读文件句柄 for fh in file_handlers: fh.close()
更简洁的Pandas实现(允许安装第三方库时使用)
import os import glob import pandas as pd root_dir = "W:/certaindirectory/" txt_subpath = "xxx/xxx/" scan_folders = [f for f in os.listdir(root_dir) if f.startswith("scan")] for folder in scan_folders: txt_dir = os.path.join(root_dir, folder, txt_subpath) txt_files = sorted(glob.glob(os.path.join(txt_dir, "*.txt"))) if not txt_files: continue # 读取第一个文件,设置列名 df = pd.read_csv(txt_files[0], sep="\t", header=None, names=["X", "Y", os.path.basename(txt_files[0])]) # 拼接剩余文件的第三列 for f in txt_files[1:]: col_name = os.path.basename(f) df[col_name] = pd.read_csv(f, sep="\t", header=None, usecols=[2]) # 输出结果 output_path = os.path.join(root_dir, folder, f"{folder}.csv") df.to_csv(output_path, index=False, encoding="utf-8")
内容的提问来源于stack exchange,提问作者quester
相关产品推荐
相关产品推荐

