You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python docx脚本仅识别单个待标记单词,请求排查循环错误

Python docx脚本仅匹配单个目标单词的问题排查与修复

我写了一段Python脚本,想从docx文档里找出to_remove列表中的单词,把它们设为加粗红色格式。脚本能运行,但只能识别列表里的第一个单词,后面的循环好像没执行。代码如下:

import docx
import tkinter as tk
from tkinter import filedialog
import os
from docx.shared import RGBColor

def replace_words(doc_name, words_dict, to_remove):

    doc = docx.Document(doc_name)
    for para in doc.paragraphs:
        for find_text, replace_text in words_dict.items():
            para.text = para.text.replace(find_text, replace_text)
        for run in para.runs:
            for to_remove_text in to_remove:
                if to_remove_text in run.text:
                    idx = run.text.find(to_remove_text)
                    if idx > 0:
                        para.add_run(run.text[:idx])
                    new_run = para.add_run(to_remove_text)
                    new_run.font.bold = True
                    new_run.font.color.rgb = RGBColor(255, 0, 0)
                    if idx + len(to_remove_text) < len(run.text):
                        para.add_run(run.text[idx + len(to_remove_text):])
                    run._element.clear_content()

    new_doc_name = os.path.splitext(doc_name)[0] + "_replaced" + os.path.splitext(doc_name)[1]
    doc.save(new_doc_name)

def browse_file():

    file_path = filedialog.askopenfilename(initialdir="/", title="Select file",
                                           filetypes=("Word documents", "*.docx"), ("all files", "*.*")))
    return file_path


root = tk.Tk()
root.withdraw()

file_path = browse_file()

if file_path:
    words_dict = {"eco-friendly": "sustainable", "nature-friendly": "natural", "ecological": "planet friendly",
                  "high quality": "excellent", "top quality": "outstanding", "premium quality": "superb",
                  "suitable for": "compatible for", "guarantee": "commitment","Eco-friendly": "Sustainable",
                  "Nature-friendly": "Natural", "Ecological": "Planet friendly",
                  "High quality": "Excellent", "Top quality": "Outstanding", "Premium quality": "Superb",
                  "Suitable for": "Compatible for", "Guarantee": "Commitment"}

    # words_dict = {key.lower(): value for key, value in words_dict.items()}

    to_remove = ["free shipping", "low price", "cheap", "buy now", "top rater", "bestseller","bestseller!"
                 "satisfaction guaranteed", "discount", "special promotion", "on sale", "top selling","Free shipping",
                 "Low price", "Cheap", "Buy now", "Top rater", "Bestseller","Bestseller!"
                 "Satisfaction guaranteed", "Discount", "Special promotion", "On sale", "Top selling"]
    # remove_list = [word.lower() for word in to_remove]

    replace_words(file_path, words_dict, to_remove)

    print("Words replaced successfully. New document created.")

else:
    print("No file selected.")

问题根源

  1. 列表语法错误:to_remove列表中多个元素缺失逗号,比如"bestseller!"与"satisfaction guaranteed"直接拼接成了一个字符串,"Bestseller!"与"Satisfaction guaranteed"同理。这导致列表实际少了多个目标单词,自然无法被匹配。
  2. Run处理逻辑缺陷:一旦在某个run中找到匹配单词,就执行run._element.clear_content()清空当前run,后续循环无法处理该run内的其他目标单词;且直接拆分run并添加新run的方式,会打乱原有run结构,无法处理单个run内多个匹配单词的场景。
  3. 函数参数语法错误:browse_file函数中filedialog.askopenfilename的filetypes参数多了一个右括号,会导致函数调用失败(属于输入笔误)。

修复后的代码

import docx
import tkinter as tk
from tkinter import filedialog
import os
from docx.shared import RGBColor

def replace_words(doc_name, words_dict, to_remove):
    doc = docx.Document(doc_name)
    for para in doc.paragraphs:
        # 先完成单词替换操作
        for find_text, replace_text in words_dict.items():
            para.text = para.text.replace(find_text, replace_text)
        
        # 重新遍历处理后的段落runs,替换会重置段落runs结构
        new_runs = []
        for run in para.runs:
            current_text = run.text
            while current_text:
                matched = False
                for to_remove_text in to_remove:
                    start_idx = current_text.find(to_remove_text)
                    if start_idx != -1:
                        # 添加匹配词之前的文本,保留原格式
                        if start_idx > 0:
                            pre_run = docx.text.run.Run(run._element, run._parent)
                            pre_run.text = current_text[:start_idx]
                            pre_run.font.bold = run.font.bold
                            pre_run.font.color.rgb = run.font.color.rgb
                            new_runs.append(pre_run)
                        # 添加加粗红色的目标单词
                        highlight_run = docx.text.run.Run(run._element, run._parent)
                        highlight_run.text = to_remove_text
                        highlight_run.font.bold = True
                        highlight_run.font.color.rgb = RGBColor(255, 0, 0)
                        new_runs.append(highlight_run)
                        # 截取剩余文本继续处理
                        current_text = current_text[start_idx + len(to_remove_text):]
                        matched = True
                        break
                # 没有匹配到则直接添加剩余文本
                if not matched:
                    new_run = docx.text.run.Run(run._element, run._parent)
                    new_run.text = current_text
                    new_run.font.bold = run.font.bold
                    new_run.font.color.rgb = run.font.color.rgb
                    new_runs.append(new_run)
                    current_text = ""
        
        # 清空原段落内容,添加处理后的新runs
        para._element.clear_content()
        for new_run in new_runs:
            para._element.append(new_run._element)

    new_doc_name = os.path.splitext(doc_name)[0] + "_replaced" + os.path.splitext(doc_name)[1]
    doc.save(new_doc_name)

def browse_file():
    # 修复filetypes参数的括号错误
    file_path = filedialog.askopenfilename(
        initialdir="/", 
        title="Select file",
        filetypes=(("Word documents", "*.docx"), ("all files", "*.*"))
    )
    return file_path

root = tk.Tk()
root.withdraw()

file_path = browse_file()

if file_path:
    words_dict = {
        "eco-friendly": "sustainable", "nature-friendly": "natural", 
        "ecological": "planet friendly", "high quality": "excellent", 
        "top quality": "outstanding", "premium quality": "superb",
        "suitable for": "compatible for", "guarantee": "commitment",
        "Eco-friendly": "Sustainable", "Nature-friendly": "Natural", 
        "Ecological": "Planet friendly", "High quality": "Excellent", 
        "Top quality": "Outstanding", "Premium quality": "Superb",
        "Suitable for": "Compatible for", "Guarantee": "Commitment"
    }

    # 修复列表逗号缺失问题,拆分拼接的字符串
    to_remove = [
        "free shipping", "low price", "cheap", "buy now", "top rater", 
        "bestseller", "bestseller!", "satisfaction guaranteed", "discount", 
        "special promotion", "on sale", "top selling",
        "Free shipping", "Low price", "Cheap", "Buy now", "Top rater", 
        "Bestseller", "Bestseller!", "Satisfaction guaranteed", "Discount", 
        "Special promotion", "On sale", "Top selling"
    ]

    replace_words(file_path, words_dict, to_remove)
    print("Words replaced successfully. New document created.")
else:
    print("No file selected.")

关键修改说明

  • 修复to_remove列表的逗号缺失问题,确保每个目标单词都是独立的列表元素。
  • 重构run处理逻辑:先拆分每个run的文本,循环处理所有匹配的目标单词,收集处理后的新run,最后替换原段落的runs,避免直接修改原run导致的逻辑混乱,支持单个run内多个匹配单词的场景。
  • 修复browse_file函数的参数语法错误。
  • 保留原run的格式(除目标单词的加粗红色),避免破坏文档原有排版。

内容的提问来源于stack exchange,提问作者Hubert Bokszczanin

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.29 19:03:02