You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Spacy提取主/宾语时完整专有名词识别问题求助

解决Spacy识别完整专有名词主/宾语的问题

问题背景

刚用Python的Spacy处理文本时碰到一个问题:Spacy无法将**完整的专有名词(比如John Doe、John Smith这类)**正确标记为主语或宾语。当语境里同时出现这类共享前缀的名字时,Spacy会把John和后面的姓氏拆分开,导致混淆Doe和Smith的归属。想知道能不能通过规则注入实现类似「若Doe紧跟John,则John Doe是主语;若Smith紧跟John,则John Smith是主语」的逻辑。

解决方案

1. 先让Spacy正确合并完整专有名词

Spacy默认分词可能会把完整人名拆成单个token,第一步要确保它把John Doe这类识别成完整的名词短语,用EntityRuler就能实现:

import spacy
from spacy.pipeline import EntityRuler

nlp = spacy.load("en_core_web_lg")
# 在NER组件前添加实体匹配器
ruler = nlp.add_pipe("entity_ruler", before="ner")

# 定义要匹配的完整人名规则
patterns = [
    {"label": "PERSON", "pattern": [{"LOWER": "john"}, {"LOWER": "doe"}]},
    {"label": "PERSON", "pattern": [{"LOWER": "john"}, {"LOWER": "smith"}]}
]
ruler.add_patterns(patterns)

# 测试效果
doc = nlp("John Doe met John Smith yesterday.")
for ent in doc.ents:
    print(ent.text, ent.label_)
# 输出:John Doe PERSON,John Smith PERSON

2. 用Spacy的Matcher实现主/宾语规则匹配

如果要精准控制主宾语的识别逻辑,用Spacy的Matcher替代正则会更适配语法结构,避免正则的局限性:

from spacy.matcher import Matcher

matcher = Matcher(nlp.vocab)

# 规则:John Doe作为主语(后跟动词)
subj_pattern = [
    {"LOWER": "john"}, {"LOWER": "doe"},
    {"POS": "VERB", "OP": "+"}
]
# 规则:John Smith作为宾语(前跟动词或介词)
obj_pattern = [
    {"POS": {"IN": ["VERB", "ADP"]}},
    {"LOWER": "john"}, {"LOWER": "smith"}
]

matcher.add("JOHN_DOE_SUBJ", [subj_pattern])
matcher.add("JOHN_SMITH_OBJ", [obj_pattern])

doc = nlp("John Doe welcomed John Smith.")
matches = matcher(doc)
for match_id, start, end in matches:
    span = doc[start:end]
    print(f"匹配结果:{span.text},类型:{nlp.vocab.strings[match_id]}")

3. 修正现有正则的问题

你当前用的正则是拆分匹配单个名字,没法处理完整人名。可以把names改成完整人名的列表,再转义特殊字符适配正则:

import re

names = ["John Doe", "John Smith"]
escaped_names = [re.escape(name) for name in names]
# 调整后的间接宾语正则
indObjCombinations = re.compile(r'\b(with|for|against|to|from|without|between)(?:\W+\w+){0,3}?\W+(%s)\b' % '|'.join(escaped_names), re.IGNORECASE)

你当前的代码

if lang == 'en':    
    dir_path = r'/User/news/articles/en'
    nlp = en_core_web_lg.load()
    # if name comes after these words he is most likely the object
    indObjCombinations = re.compile(r'\b(with|for|against|to|from|without|between)(?:\W+\w+){0,3}?\W+(%s)\b' % '|'.join(names),re.IGNORECASE)
    passiveCombinations = re.compile(r'\b(by)(?:\W+\w+){0,3}?\W+(%s)\b' % '|'.join(names),re.IGNORECASE)
    objCombinations =  re.compile(r'\b(received|receives|welcomes|welcomed)(?:\W+\w+){0,3}?\W+(%s)\b' % '|'.join(names),re.IGNORECASE)

    subjCombinations = re.compile(r'\b(%s)(?:\W+\w+){0,1}?\W+(in)\b' % '|'.join(names),re.IGNORECASE)
    subjCombinations_andName = re.compile(r'\b(and)(\W+\w+){0,3}\W+(%s)(\W+\w+){0,3}\W+(are)\b' % '|'.join(names),re.IGNORECASE )
    subjCombinations_nameAnd = re.compile(r'\b(%s)(\W+\w+){0,3}\W+(and)(\W+\w+){0,3}\W+(are)\b' % '|'.join(names),re.IGNORECASE )
    subjBeginning = re.compile(r'\b^(%s)(\W+\w+){0,1}\W*(:)\b' % '|'.join(names),re.IGNORECASE )


def getSubsFromConjunctions(subs):
    moreSubs = []
    for sub in subs:
        # rights is a generator
        rights = list(sub.rights)
        rightDeps = {tok.lower_ for tok in rights}

        if lang == 'en':
            if "and" in rightDeps:
                moreSubs.extend([tok for tok in rights if tok.dep_ in SUBJECTS or tok.pos_ == "NOUN"])
                if len(moreSubs) > 0:
                    moreSubs.extend(getSubsFromConjunctions(moreSubs))

def getObjsFromConjunctions(objs):
    moreObjs = []
    for obj in objs:
        # rights is a generator
        rights = list(obj.rights)
        rightDeps = {tok.lower_ for tok in rights}
        if lang == 'en':
            if "and" in rightDeps:
                moreObjs.extend([tok for tok in rights if tok.dep_ in OBJECTS or tok.pos_ == "NOUN"])
                if len(moreObjs) > 0:
                     moreObjs.extend(getObjsFromConjunctions(moreObjs))

内容的提问来源于stack exchange,提问作者Mark Noel

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.16 21:31:11