You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python批量处理TXT文件提取数据并导出CSV的技术咨询

批量处理TXT文件并导出CSV的问题解决方案

我写了一段Python代码批量处理文件夹里的TXT文件,已经能定位以"Read"开头但排除"Read Segment"等特定变体的行,但卡在两个问题上:

  1. 字符串解析逻辑该放在代码的哪个位置?
  2. 怎么把解析出的年份、数值数据导出成指定格式的CSV?

现有代码、样本TXT内容及理想CSV格式如下:

import os
import argparse
import csv
from typing import List


def validate_directory(path):
    if os.path.isdir(path):
        return path
    else:
        raise NotADirectoryError(path)


def get_data_from_file(file) -> List[str]:
    ignore_list = ["Read Segment", "Read Disk", "Read a line", "Read in"]
    data = []
    with open(file, "r", encoding="latin1") as f:
        try:
            lines = f.readlines()
        except Exception as e:
            print(f"Unable to process {file}: {e}")
            return []
        for line_number, line in enumerate(lines, start=1):
            if not any(variation in line for variation in ignore_list):
                if line.strip().startswith("Read ") and not line.strip().startswith("Read ("): # TODO: fix this with better regex
                    data.append(f'Found "Read" at line {line_number} in {file}')
                    print(f'Found "Read" at {file}:{line_number}')
                    print(lines[line_number-1])
    return data


def list_read_data(directory_path: str) -> List[str]:
    total_data = []
    for root, _, files in os.walk(directory_path):
        for file_name in files:
            if file_name.endswith(".txt"):
                data = get_data_from_file(os.path.join(root, file_name))
                total_data.extend(data)

    return total_data


def write_results_to_csv(output_file: str, data: List[str]):
    with open(output_file, "w", newline="", encoding="utf-8") as csvfile:
        writer = csv.writer(csvfile)
        writer.writerow(["Results"])
        for line in data:
            writer.writerow([line])


def main(directory_path: str, output_file: str):
    data = list_read_data(directory_path)
    write_results_to_csv(output_file, data)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(
        description="Process the 2020Model folder for input data."
    )
    parser.add_argument(
        "--directory", type=validate_directory, help="folder to be processed"
    )
    parser.add_argument("--output", type=str, help="Output file name (e.g., outputfile.csv)", default="outputfile.csv")

    args = parser.parse_args()
    main(os.path.abspath(args.directory), args.output)

样本TXT内容:

Select Year(2007-2025)
Read TotPkSav
/2007     2008     2009     2010     2011     2012     2013     2014     2015     2016     2017     2018     2019     2020     2021     2022     2023     2024     2025 
   00       27       53       78      108      133      151      161      169      177      186      195      205      216      229      242      257      273      288 

理想CSV格式:

1985,1986,1986,1987,1988,1989,1990,1991,1992,1993,1994
37839,36962,37856,41971,40838,44640.87,42826.34,44883.03,43077.59,45006.49,46789

问题1:字符串解析逻辑的位置

解析逻辑应该直接放在get_data_from_file函数内部——就在你定位到目标"Read"行之后。这个函数的核心职责是读取单个文件并提取有效数据,解析紧邻"Read"行的年份、数值行是它的自然延伸,这样代码逻辑更连贯,也方便后续维护。

问题2:导出指定格式的CSV

需要调整数据存储结构:不再存字符串提示,而是存储年份列表和对应的数值列表;然后在写入CSV时,先写年份行,再写数值行,就能得到你要的格式。

修改后的完整代码

import os
import argparse
import csv
import re
from typing import List, Tuple


def validate_directory(path):
    if os.path.isdir(path):
        return path
    else:
        raise NotADirectoryError(path)


def get_data_from_file(file) -> List[Tuple[List[str], List[float]]]:
    # 用正则精准匹配需要排除的Read变体
    ignore_patterns = r"Read (Segment|Disk|a line|in)"
    data = []
    with open(file, "r", encoding="latin1") as f:
        try:
            # 过滤空行并去除首尾空白,方便后续处理
            lines = [line.strip() for line in f.readlines() if line.strip()]
        except Exception as e:
            print(f"无法处理文件 {file}: {e}")
            return []
        
        for idx, line in enumerate(lines):
            # 跳过需要忽略的Read变体
            if re.search(ignore_patterns, line):
                continue
            # 匹配合法的Read行
            if line.startswith("Read ") and not line.startswith("Read ("):
                # 检查后续是否有年份行(以/开头)和数值行
                if idx + 1 < len(lines) and lines[idx+1].startswith("/"):
                    # 解析年份:去掉开头的/,分割成列表
                    year_line = lines[idx+1].lstrip("/")
                    years = [y.strip() for y in year_line.split() if y.strip()]
                    # 解析数值:分割后转成float,处理"00"这类特殊值
                    if idx + 2 < len(lines):
                        value_line = lines[idx+2]
                        values = []
                        for val in value_line.split():
                            try:
                                values.append(float(val) if val != "00" else 0.0)
                            except ValueError:
                                values.append(0.0)  # 异常值默认设为0
                        data.append((years, values))
                        print(f"在文件 {file} 中解析到一组数据:{len(years)}个年份,{len(values)}个数值")
    return data


def list_read_data(directory_path: str) -> List[Tuple[List[str], List[float]]]:
    total_data = []
    for root, _, files in os.walk(directory_path):
        for file_name in files:
            if file_name.endswith(".txt"):
                file_path = os.path.join(root, file_name)
                data = get_data_from_file(file_path)
                total_data.extend(data)
    return total_data


def write_results_to_csv(output_file: str, data: List[Tuple[List[str], List[float]]]):
    with open(output_file, "w", newline="", encoding="utf-8") as csvfile:
        writer = csv.writer(csvfile)
        # 逐组写入:先写年份行,再写数值行
        for years, values in data:
            # 补全年份和数值的长度,避免行列不对应
            max_len = max(len(years), len(values))
            padded_years = years + [""]*(max_len - len(years))
            padded_values = values + [0.0]*(max_len - len(values))
            writer.writerow(padded_years)
            writer.writerow(padded_values)


def main(directory_path: str, output_file: str):
    data = list_read_data(directory_path)
    write_results_to_csv(output_file, data)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(
        description="批量处理文件夹中的TXT文件,提取年份和数值数据并导出为CSV。"
    )
    parser.add_argument(
        "--directory", type=validate_directory, required=True, help="要处理的文件夹路径"
    )
    parser.add_argument("--output", type=str, help="输出CSV文件名", default="output.csv")

    args = parser.parse_args()
    main(os.path.abspath(args.directory), args.output)

关键修改说明

  1. 正则优化Read行匹配:用正则表达式替代原有的多条件判断,更简洁精准,也方便后续扩展忽略的变体。
  2. 解析逻辑嵌入文件读取函数:定位到Read行后直接解析后续的年份和数值行,减少数据传递的复杂度。
  3. 数据结构调整:返回(年份列表, 数值列表)的元组,让数据更结构化,便于CSV写入。
  4. CSV写入逻辑优化:自动补全年份和数值的长度,确保每行数据对齐,输出格式完全符合需求。

实际输出效果

处理你提供的样本TXT后,CSV内容为:

2007,2008,2009,2010,2011,2012,2013,2014,2015,2016,2017,2018,2019,2020,2021,2022,2023,2024,2025
0.0,27.0,53.0,78.0,108.0,133.0,151.0,161.0,169.0,177.0,186.0,195.0,205.0,216.0,229.0,242.0,257.0,273.0,288.0

用表格工具打开后,就是你想要的行列结构。

内容的提问来源于stack exchange,提问作者mm mm

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.15 00:24:51