You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Spacy分句边界检测排除下,注释化docx语料库的注释归属匹配问询

解决Spacy分句时排除批注占位符干扰的实用方案

我之前处理过类似带自定义标记的文本分句场景,给你几个落地性强的方案,帮你避开<|n|>/<||n||>这类占位符对Spacy分句边界检测的干扰:


方案1:先替换占位符,分句后再还原(最省心的思路)

核心是先把特殊占位符换成Spacy分句器不会误判的临时标记,分句完成后再还原回去,完全绕开干扰问题。

import re
import spacy

# 加载Spacy模型
nlp = spacy.load("en_core_web_sm")

# 1. 定义占位符替换/还原逻辑
def replace_placeholders(text):
    # 匹配两种占位符格式,同时记录是开始还是结束标记
    pattern = re.compile(r'(<\|(\d+)\|>)|(<\|\|(\d+)\|\|>)')
    def replacer(match):
        if match.group(1):
            # 开始标记换成§START_n§
            return f'§START_{match.group(2)}§'
        else:
            # 结束标记换成§END_n§
            return f'§END_{match.group(4)}§'
    return pattern.sub(replacer, text)

def restore_placeholders(sentences):
    restored_sents = []
    for sent in sentences:
        # 把临时标记还原回原始占位符
        restored = re.sub(r'§START_(\d+)§', r'<|\1|>', sent.text)
        restored = re.sub(r'§END_(\d+)§', r'<||\1||>', restored)
        restored_sents.append(restored)
    return restored_sents

# 2. 实际处理流程
raw_text = "Lorem ipsum <|1|> dolor sit amet, consectetur <||1||> adipiscing elit. Sed do eiusmod <|2|> tempor incididunt <||2||> ut labore et dolore magna aliqua."

# 替换占位符
processed_text = replace_placeholders(raw_text)
# 分句
doc = nlp(processed_text)
# 还原占位符
final_sentences = restore_placeholders(doc.sents)

# 输出结果
for idx, sent in enumerate(final_sentences, 1):
    print(f"句子{idx}: {sent}")

方案2:自定义Spacy分词&分句规则(直接修改Spacy行为)

如果不想做预处理,可以直接调整Spacy的分词器和分句器,让它把占位符当成完整token,并且不触发分句边界。

import spacy
from spacy.language import Language
from spacy.tokens import Doc

# 加载模型
nlp = spacy.load("en_core_web_sm")

# 第一步:修改分词器,让占位符成为完整token
# 移除<作为前缀、>作为后缀,避免被拆分
prefixes = list(nlp.Defaults.prefixes)
prefixes.remove('<')
nlp.tokenizer.prefix_search = spacy.util.compile_prefix_regex(prefixes).search

suffixes = list(nlp.Defaults.suffixes)
suffixes.remove('>')
nlp.tokenizer.suffix_search = spacy.util.compile_suffix_regex(suffixes).search

# 移除|作为分隔符,避免占位符内部被拆分
infixes = [x for x in nlp.Defaults.infixes if '|' not in x]
nlp.tokenizer.infix_finditer = spacy.util.compile_infix_regex(infixes).finditer

# 第二步:自定义分句器,忽略占位符的分句触发
@Language.component("ignore_annot_placeholders")
def ignore_annot_placeholders(doc):
    new_sents = []
    current_sent_tokens = []
    
    for token in doc:
        # 判断是否是批注占位符
        is_placeholder = token.text.startswith('<|') and token.text.endswith('|>') or \
                         token.text.startswith('<||') and token.text.endswith('||>')
        
        current_sent_tokens.append(token)
        # 只有当非占位符token触发分句时,才分割句子
        if token.is_sent_end and not is_placeholder:
            new_sents.append(Doc(doc.vocab, [t.text for t in current_sent_tokens]))
            current_sent_tokens = []
    
    # 处理最后一段未闭合的句子
    if current_sent_tokens:
        new_sents.append(Doc(doc.vocab, [t.text for t in current_sent_tokens]))
    
    doc.sents = new_sents
    return doc

# 添加自定义分句器到管道
nlp.add_pipe("ignore_annot_placeholders", before="parser")

# 测试
test_text = "Lorem ipsum <|1|> dolor sit amet, consectetur <||1||> adipiscing elit. Sed do eiusmod <|2|> tempor incididunt <||2||> ut labore et dolore magna aliqua."
doc = nlp(test_text)
for sent in doc.sents:
    print(sent.text)

后续:匹配注释所属句子的小技巧

分句完成后,你可以通过索引对比的方式,把每个注释绑定到对应的句子:

import re

# 提取所有注释的起止位置和ID
annot_pattern = re.compile(r'<\|(\d+)\|>(.*?)<\|\|\1\|\|>', re.DOTALL)
annotations = []
for match in annot_pattern.finditer(raw_text):
    annotations.append({
        "id": match.group(1),
        "start": match.start(),
        "end": match.end(),
        "content": match.group(2)
    })

# 获取每个句子的起止索引
sentence_ranges = [(sent.start_char, sent.end_char, sent.text) for sent in nlp(raw_text)]

# 匹配注释和句子
for annot in annotations:
    for sent_start, sent_end, sent_text in sentence_ranges:
        if annot["start"] >= sent_start and annot["end"] <= sent_end:
            print(f"注释ID {annot['id']} 属于句子:{sent_text}")
            break
    else:
        print(f"注释ID {annot['id']} 跨多个句子,需特殊处理")

内容的提问来源于stack exchange,提问作者neo

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.22 08:57:05