You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python实现:文本文件下一行不以|开头时替换对应换行符

Python文本批量处理实现方案

方案说明

提供两种实现方式,按行处理逻辑清晰易修改,正则实现和你之前Notepad++的操作逻辑一致,两种都完全匹配你给出的示例要求。

完整代码

按行处理版本(推荐)

import os

def process_single_file(input_path, output_path, encoding='utf-8'):
    # 读取所有行并去除末尾换行符
    with open(input_path, 'r', encoding=encoding) as f:
        lines = [line.rstrip('\n').rstrip('\r') for line in f.readlines()]
    
    result = []
    current_block = None
    
    for line in lines:
        if line.startswith('|'):
            # 拼接前一个块的内容加入结果
            if current_block is not None:
                extra_len = len(current_block['extra'])
                if extra_len == 1:
                    block_str = f"{current_block['header']}|{current_block['extra'][0]}"
                elif extra_len >= 2:
                    block_str = f"{current_block['header']}|{'|'.join(current_block['extra'][:-1])}\\n{current_block['extra'][-1]}"
                result.append(block_str)
            # 初始化新块
            current_block = {
                'header': line,
                'extra': []
            }
        else:
            # 非|开头的行加入当前块
            if current_block is not None:
                current_block['extra'].append(line)
    
    # 处理最后一个未拼接的块
    if current_block is not None:
        extra_len = len(current_block['extra'])
        if extra_len == 1:
            block_str = f"{current_block['header']}|{current_block['extra'][0]}"
        elif extra_len >= 2:
            block_str = f"{current_block['header']}|{'|'.join(current_block['extra'][:-1])}\\n{current_block['extra'][-1]}"
        result.append(block_str)
    
    # 写入结果文件
    with open(output_path, 'w', encoding=encoding) as f:
        f.write('\n'.join(result))

def batch_process(input_dir, output_dir, encoding='utf-8'):
    # 自动创建输出目录
    if not os.path.exists(output_dir):
        os.makedirs(output_dir)
    # 遍历目录下所有txt文件处理,可修改后缀匹配规则
    for filename in os.listdir(input_dir):
        if filename.endswith('.txt'):
            input_path = os.path.join(input_dir, filename)
            output_path = os.path.join(output_dir, filename)
            process_single_file(input_path, output_path, encoding)
            print(f"已完成:{filename}")

if __name__ == '__main__':
    # 单个文件处理调用方式
    # process_single_file('输入文件路径.txt', '输出文件路径.txt')
    # 批量处理调用方式
    batch_process('待处理文件目录路径', '处理后文件存储目录路径')

正则实现版本

import os
import re

def process_single_file_regex(input_path, output_path, encoding='utf-8'):
    with open(input_path, 'r', encoding=encoding) as f:
        content = f.read()
    # 第一步:替换块首行和首个非|行之间的换行符为|
    content = re.sub(r'(\|.*?)(\r?\n)(?!\|)', r'\1|', content, flags=re.MULTILINE)
    # 第二步:替换非|行之间的换行符为字面量\n
    content = re.sub(r'(\r?\n)(?!\|)', r'\\n', content, flags=re.MULTILINE)
    with open(output_path, 'w', encoding=encoding) as f:
        f.write(content)

# 批量处理逻辑和上面按行处理版本一致,直接复用batch_process函数即可

使用说明

  1. 批量处理时,先把所有待处理的txt文件放到同一个目录下,修改代码里的待处理文件目录路径和处理后文件存储目录路径后运行即可
  2. 如果需要处理其他后缀的文件,修改filename.endswith('.txt')里的后缀即可
  3. 读写编码默认用utf-8,如果需要兼容gbk等编码,修改调用时的encoding参数即可

内容的提问来源于stack exchange,提问作者Hannibal

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.10.06 08:42:03