如何用BeautifulSoup或正则替换HTML指定CSS选择器下的任意子树
实现思路与可运行代码
核心逻辑
- 首先区分两种匹配场景:
old为纯文本、old为带结构的HTML子树 - 所有匹配仅作用于DOM节点的内容部分,完全跳过标签属性、注释等非内容区域,避免误替换
- 优先实现普适性最高的纯文本匹配场景,可直接覆盖你给出的测试用例需求;子树匹配逻辑预留扩展空间,可根据实际需求调整匹配精度
完整实现代码
import bs4 from bs4 import BeautifulSoup, NavigableString, Comment def replace(html: str, selector: str, old: str, new: str) -> str: # 统一用html.parser解析,避免额外依赖,也可替换为lxml提升性能 soup = BeautifulSoup(html, "html.parser") old_soup = BeautifulSoup(old, "html.parser") new_soup = BeautifulSoup(new, "html.parser") # 提取old的实际内容,判断是否为纯文本匹配模式 old_content = old_soup.body.contents is_plain_text_old = len(old_content) == 1 and isinstance(old_content[0], NavigableString) old_text = old_content[0] if is_plain_text_old else None new_content = new_soup.body.contents for selected_node in soup.select(selector): if is_plain_text_old: # 纯文本匹配模式:遍历所有文本节点,跳过注释 for text_node in list(selected_node.descendants): if isinstance(text_node, Comment) or not isinstance(text_node, NavigableString): continue if old_text not in text_node: continue # 拆分原文本,插入新内容替换匹配项 parent = text_node.parent parts = text_node.split(old_text) text_node.extract() # 按拆分结果插入文本和替换后的节点 for idx, part in enumerate(parts): if part: parent.insert(idx * 2, part) if idx != len(parts) - 1: # 节点拷贝避免同一对象复用异常 for new_node in [n.copy() for n in new_content]: parent.insert(idx * 2 + 1, new_node) else: # HTML子树匹配模式:递归对比节点结构、属性、内容 def match_subtree(node, target_subtree): if len(node.contents) != len(target_subtree): return False for n1, n2 in zip(node.contents, target_subtree): if isinstance(n1, NavigableString) and isinstance(n2, NavigableString): if n1 != n2: return False elif n1.name != n2.name or n1.attrs != n2.attrs: return False elif not match_subtree(n1, n2.contents): return False return True for child in list(selected_node.descendants): if not hasattr(child, 'contents'): continue if match_subtree(child, old_content): child.replace_with(*[n.copy() for n in new_content]) return str(soup)
测试验证
使用你给出的测试用例可直接验证符合预期:
before = r"""<!DOCTYPE html> <html> <head> <title>No target here</title> </head> <body> <h1>This is the target!</h1> <p class="target"> Yet another <b>target</b>. </p> <p> <!-- Comment --> Foo target Bar </p> </body> </html> """ after = replace( html = before, selector = 'body', old = 'target', new = '<span class="special">target</span>', ) print(after)
内容的提问来源于stack exchange,提问作者s-m-e
相关产品推荐
相关产品推荐

