如何使用Stanza/Spacy判断两个指定关键词是否存在依存句法关系
两个短语依存关系判断代码实现
前置逻辑说明
- 首先从解析后的doc对象中匹配到两个目标短语对应的token列表
- 遍历两个短语的所有token组合,判断是否存在依存关联:可根据需求选择仅判断直接父子依存,或扩展为判断任意依存路径上的间接关联
spaCy 实现代码
import spacy from spacy.matcher import Matcher # 初始化模型和输入数据 nlp = spacy.load("en_core_web_sm") doc = nlp("His home was in violation of local and state zoning and environmental regulations, and there was no access to a road") keyword = {"changes": ["no access"], "features": ["road"]} phrase1 = keyword["changes"][0] phrase2 = keyword["features"][0] def get_phrase_span(doc, target_phrase): """匹配目标短语在doc中对应的Span对象""" matcher = Matcher(nlp.vocab) pattern = [{"LOWER": token.text.lower()} for token in nlp(target_phrase)] matcher.add("PHRASE_MATCH", [pattern]) matches = matcher(doc) for match_id, start, end in matches: return doc[start:end] return None def check_phrase_dependency(doc, phrase_a, phrase_b, check_indirect=True): span_a = get_phrase_span(doc, phrase_a) span_b = get_phrase_span(doc, phrase_b) if not span_a or not span_b: return False tokens_b = set(span_b) tokens_a = set(span_a) for token_a in span_a: # 检查直接依存 if token_a.head in tokens_b: return True # 检查间接依存(任意依存路径关联) if check_indirect: for ancestor in token_a.ancestors: if ancestor in tokens_b: return True for token_b in span_b: if token_b.head in tokens_a: return True if check_indirect: for ancestor in token_b.ancestors: if ancestor in tokens_a: return True return False # 调用测试,默认开启间接依存检查,符合示例场景会返回True print(check_phrase_dependency(doc, phrase1, phrase2))
Stanza 实现代码
import stanza # 初始化模型和输入数据 nlp = stanza.Pipeline(lang='en', processors='tokenize,pos,lemma,depparse', download_method=None) doc = nlp("His home was in violation of local and state zoning and environmental regulations, and there was no access to a road") keyword = {"changes": ["no access"], "features": ["road"]} phrase1 = keyword["changes"][0] phrase2 = keyword["features"][0] def get_phrase_tokens(doc, target_phrase): """匹配目标短语在doc中对应的token列表""" target_tokens = [t.text.lower() for t in nlp(target_phrase).sentences[0].words] target_len = len(target_tokens) for sent in doc.sentences: words = sent.words for i in range(len(words) - target_len + 1): if [w.text.lower() for w in words[i:i+target_len]] == target_tokens: return words[i:i+target_len] return None def check_phrase_dependency(doc, phrase_a, phrase_b, check_indirect=True): tokens_a = get_phrase_tokens(doc, phrase_a) tokens_b = get_phrase_tokens(doc, phrase_b) if not tokens_a or not tokens_b: return False token_a_ids = set([t.id for t in tokens_a]) token_b_ids = set([t.id for t in tokens_b]) sent = doc.sentences[0] for t_a in tokens_a: # 检查直接依存 if t_a.head in token_b_ids: return True # 检查间接依存 if check_indirect: current = t_a while current.head != 0: current = sent.words[current.head - 1] if current.id in token_b_ids: return True for t_b in tokens_b: if t_b.head in token_a_ids: return True if check_indirect: current = t_b while current.head != 0: current = sent.words[current.head - 1] if current.id in token_a_ids: return True return False # 调用测试,默认开启间接依存检查,符合示例场景会返回True print(check_phrase_dependency(doc, phrase1, phrase2))
内容的提问来源于stack exchange,提问作者Sangeeth Saseendran Nair
相关产品推荐
相关产品推荐

