Python实现:文本文件下一行不以|开头时替换对应换行符
Python文本批量处理实现方案
方案说明
提供两种实现方式,按行处理逻辑清晰易修改,正则实现和你之前Notepad++的操作逻辑一致,两种都完全匹配你给出的示例要求。
完整代码
按行处理版本(推荐)
import os def process_single_file(input_path, output_path, encoding='utf-8'): # 读取所有行并去除末尾换行符 with open(input_path, 'r', encoding=encoding) as f: lines = [line.rstrip('\n').rstrip('\r') for line in f.readlines()] result = [] current_block = None for line in lines: if line.startswith('|'): # 拼接前一个块的内容加入结果 if current_block is not None: extra_len = len(current_block['extra']) if extra_len == 1: block_str = f"{current_block['header']}|{current_block['extra'][0]}" elif extra_len >= 2: block_str = f"{current_block['header']}|{'|'.join(current_block['extra'][:-1])}\\n{current_block['extra'][-1]}" result.append(block_str) # 初始化新块 current_block = { 'header': line, 'extra': [] } else: # 非|开头的行加入当前块 if current_block is not None: current_block['extra'].append(line) # 处理最后一个未拼接的块 if current_block is not None: extra_len = len(current_block['extra']) if extra_len == 1: block_str = f"{current_block['header']}|{current_block['extra'][0]}" elif extra_len >= 2: block_str = f"{current_block['header']}|{'|'.join(current_block['extra'][:-1])}\\n{current_block['extra'][-1]}" result.append(block_str) # 写入结果文件 with open(output_path, 'w', encoding=encoding) as f: f.write('\n'.join(result)) def batch_process(input_dir, output_dir, encoding='utf-8'): # 自动创建输出目录 if not os.path.exists(output_dir): os.makedirs(output_dir) # 遍历目录下所有txt文件处理,可修改后缀匹配规则 for filename in os.listdir(input_dir): if filename.endswith('.txt'): input_path = os.path.join(input_dir, filename) output_path = os.path.join(output_dir, filename) process_single_file(input_path, output_path, encoding) print(f"已完成:{filename}") if __name__ == '__main__': # 单个文件处理调用方式 # process_single_file('输入文件路径.txt', '输出文件路径.txt') # 批量处理调用方式 batch_process('待处理文件目录路径', '处理后文件存储目录路径')
正则实现版本
import os import re def process_single_file_regex(input_path, output_path, encoding='utf-8'): with open(input_path, 'r', encoding=encoding) as f: content = f.read() # 第一步:替换块首行和首个非|行之间的换行符为| content = re.sub(r'(\|.*?)(\r?\n)(?!\|)', r'\1|', content, flags=re.MULTILINE) # 第二步:替换非|行之间的换行符为字面量\n content = re.sub(r'(\r?\n)(?!\|)', r'\\n', content, flags=re.MULTILINE) with open(output_path, 'w', encoding=encoding) as f: f.write(content) # 批量处理逻辑和上面按行处理版本一致,直接复用batch_process函数即可
使用说明
- 批量处理时,先把所有待处理的txt文件放到同一个目录下,修改代码里的
待处理文件目录路径和处理后文件存储目录路径后运行即可 - 如果需要处理其他后缀的文件,修改
filename.endswith('.txt')里的后缀即可 - 读写编码默认用utf-8,如果需要兼容gbk等编码,修改调用时的encoding参数即可
内容的提问来源于stack exchange,提问作者Hannibal
相关产品推荐
相关产品推荐

