You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何快速批量移除CSV文件中的罗马数字?现有Python方案需优化

优化CSV文件罗马数字/序号移除效率

当前使用以下代码移除CSV文件中的罗马数字及序号,但处理速度过慢:

import os

def inplace_change(filename, old_string, new_string):
    # Safely read the input filename using 'with'
    with open(filename) as f:
        s = f.read()
        if old_string not in s:
            print('"{old_string}" not found in {filename}.'.format(**locals()))
            return

    # Safely write the changed content, if found in the file
    with open(filename, 'w') as f:
        print('Changing "{old_string}" to "{new_string}" in {filename}'.format(**locals()))
        s = s.replace(old_string, new_string)
        f.write(s)

d_list = ['1. ', '2. ', '3. ', '4. ','XVIII. ','XVII. ','XVI. ','XV. ','XIV. ', 'XIII. ', 
          'XII. ','XI. ', 'IX. ','VIII. ', 'VII. ', 'VI. ','IV. ', 'IV. ', 'XVIII.','XVII.','XVI.','XV.','XIV.', 'XIII.', 
          'XII.','XI.', 'IX.','VIII.', 'VII.', 'VI.','IV.', 'IV.', 'Ⅰ.', 'Ⅱ.','Ⅲ.','Ⅳ.','Ⅴ.','Ⅵ.','Ⅶ-1.','Ⅶ-2.','Ⅶ.','Ⅱ.'
           'Ⅷ.','Ⅸ.','Ⅹ.','1.','2.','3.','4.','5.','6.','7.',
         'I. ','II. ','III. ','Ⅷ.',
         'ⅥI. ',  'VIIII. ',  '- ',  'I',  'II',
          'V.',  'Ⅵ',  'VIII',  'I.',  'II.',
          'V.',  'X.',  'Ⅹ',  'V',  'Ⅷ.',
         ]
for file in os.listdir(output_path + '/CIS'): 
    for dlist in d_list:    
        inplace_change(output_path +'/CIS/'+ file,  old_string= dlist, new_string= '')  
        continue

原代码效率低的核心原因

  • 重复IO操作:每个文件要被打开、读写数十次(d_list有多少项就操作多少次),磁盘IO是性能瓶颈
  • 重复项无用功:d_list里存在大量重复字符串(比如IV.出现多次),做了重复替换
  • 逐个替换效率低:多次调用str.replace比一次性批量替换的开销大得多

优化后的实现方案

import os
import re

def remove_target_patterns(file_path):
    # 去重并整理目标模式列表
    target_patterns = {
        '1. ', '2. ', '3. ', '4. ', '5. ', '6. ', '7.',
        'XVIII.', 'XVII.', 'XVI.', 'XV.', 'XIV.', 'XIII.', 'XII.', 'XI.', 'IX.', 'VIII.', 'VII.', 'VI.', 'IV.',
        'Ⅰ.', 'Ⅱ.', 'Ⅲ.', 'Ⅳ.', 'Ⅴ.', 'Ⅵ.', 'Ⅶ-1.', 'Ⅶ-2.', 'Ⅶ.', 'Ⅷ.', 'Ⅸ.', 'Ⅹ.',
        'I. ', 'II. ', 'III. ', 'ⅥI. ', 'VIIII. ', '- ', 'I', 'II', 'V.', 'Ⅵ', 'VIII', 'X.', 'Ⅹ', 'V'
    }
    # 转成正则表达式,转义特殊字符后用|分隔所有模式
    regex_pattern = re.compile('|'.join(re.escape(pattern) for pattern in target_patterns))
    
    # 一次性读写文件,仅操作一次IO
    with open(file_path, 'r', encoding='utf-8') as f:
        content = f.read()
    
    # 批量替换所有匹配项
    new_content = regex_pattern.sub('', content)
    
    with open(file_path, 'w', encoding='utf-8') as f:
        f.write(new_content)
    print(f"处理完成: {file_path}")

# 遍历处理所有CSV文件
cis_dir = os.path.join(output_path, 'CIS')
for filename in os.listdir(cis_dir):
    file_path = os.path.join(cis_dir, filename)
    # 仅处理CSV文件,避免误操作其他文件
    if filename.endswith('.csv'):
        remove_target_patterns(file_path)

优化点说明

  1. 去重目标模式:将d_list转成集合自动去重,避免重复替换相同字符串
  2. 正则批量替换:用正则表达式一次性匹配所有目标模式,一次替换完成所有操作,比多次str.replace效率提升明显
  3. 减少IO操作:每个文件只进行一次读、一次写,彻底解决重复IO的性能问题
  4. 增加文件过滤:只处理.csv后缀的文件,避免误处理目录下的其他文件
  5. 编码明确:指定utf-8编码,避免不同系统下的编码兼容问题

如果处理超大CSV文件(内存无法一次性容纳),可以改用按行处理的版本:

def remove_target_patterns_large_file(file_path):
    target_patterns = {
        # 同上的模式集合
        '1. ', '2. ', '3. ', '4. ', '5. ', '6. ', '7.',
        'XVIII.', 'XVII.', 'XVI.', 'XV.', 'XIV.', 'XIII.', 'XII.', 'XI.', 'IX.', 'VIII.', 'VII.', 'VI.', 'IV.',
        'Ⅰ.', 'Ⅱ.', 'Ⅲ.', 'Ⅳ.', 'Ⅴ.', 'Ⅵ.', 'Ⅶ-1.', 'Ⅶ-2.', 'Ⅶ.', 'Ⅷ.', 'Ⅸ.', 'Ⅹ.',
        'I. ', 'II. ', 'III. ', 'ⅥI. ', 'VIIII. ', '- ', 'I', 'II', 'V.', 'Ⅵ', 'VIII', 'X.', 'Ⅹ', 'V'
    }
    regex_pattern = re.compile('|'.join(re.escape(pattern) for pattern in target_patterns))
    
    # 用临时文件存储处理后的内容,避免覆盖原文件时出错
    temp_file = file_path + '.tmp'
    with open(file_path, 'r', encoding='utf-8') as f_in, open(temp_file, 'w', encoding='utf-8') as f_out:
        for line in f_in:
            new_line = regex_pattern.sub('', line)
            f_out.write(new_line)
    # 替换原文件
    os.replace(temp_file, file_path)
    print(f"处理完成大文件: {file_path}")

内容的提问来源于stack exchange,提问作者monkey821

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.19 15:55:17