如何用Spacy提取含连词与指代的法/英文NOUN-ADJ对?
提取含连词与指代关系的NOUN-ADJ对(Spacy 英语+法语实现方案)
英语实现方案
要提取包含连词(如and/but)和指代关系的NOUN-ADJ对,需依赖Spacy的依存分析和指代消解功能,注意必须使用带Transformer的英语模型(如en_coreference_web_trf),基础模型不支持指代消解。
步骤与代码示例
import spacy # 加载带指代消解的英语Transformer模型 nlp = spacy.load("en_coreference_web_trf") def extract_noun_adj_pairs(doc): pairs = set() # 用集合自动去重 # 处理指代链:将指代代词替换为原名词短语 for coref_cluster in doc.spans.get("coref", []): main_noun_span = coref_cluster[0] for ref_span in coref_cluster[1:]: ref_span._.coref_resolved = main_noun_span.text # 遍历所有名词短语 for noun_chunk in doc.noun_chunks: noun_root = noun_chunk.root # 提取直接修饰名词的形容词(匹配amod依存关系) adj_modifiers = [tok.text for tok in noun_root.children if tok.pos_ == "ADJ" and tok.dep_ == "amod"] # 添加当前名词-形容词对 for adj in adj_modifiers: pairs.add((noun_chunk.text, adj)) # 处理指代关系:如果当前名词是指代对象,关联原名词的形容词 if hasattr(noun_root, "_") and noun_root._.coref_resolved: resolved_noun = noun_root._.coref_resolved # 定位原名词在文本中的跨度 resolved_span = doc.char_span(doc.text.find(resolved_noun), doc.text.find(resolved_noun) + len(resolved_noun)) if resolved_span: resolved_adj = [tok.text for tok in resolved_span.root.children if tok.pos_ == "ADJ" and tok.dep_ == "amod"] for adj in resolved_adj: pairs.add((resolved_noun, adj)) # 处理连词连接的并列名词:共享或各自的形容词 for conj_noun in noun_root.conjuncts: if conj_noun.pos_ != "NOUN": continue conj_span = doc.char_span(doc.text.find(conj_noun.text), doc.text.find(conj_noun.text) + len(conj_noun.text)) if not conj_span: continue # 提取并列名词自身的形容词 conj_adj = [tok.text for tok in conj_span.root.children if tok.pos_ == "ADJ" and tok.dep_ == "amod"] # 共享原名词的形容词 for adj in adj_modifiers: pairs.add((conj_noun.text, adj)) # 添加并列名词自身的形容词对 for adj in conj_adj: pairs.add((conj_noun.text, adj)) return list(pairs) # 测试 text = "The big and red cat is cute. It sits next to the small dog, which is lazy." doc = nlp(text) print(extract_noun_adj_pairs(doc)) # 输出示例:[('The big and red cat', 'big'), ('The big and red cat', 'red'), ('the small dog', 'small'), ('The big and red cat', 'cute'), ('the small dog', 'lazy')]
关键说明
- 指代消解通过
doc.spans["coref"]获取指代簇,将代词替换为原名词短语。 - 连词处理依赖
conjuncts属性,获取并列的名词,实现形容词共享逻辑。 - 仅提取
amod(属性修饰)关系的形容词,可根据需求扩展其他依存标签(如acl)覆盖更多修饰场景。
法语实现方案
法语的实现逻辑与英语一致,需使用Spacy的法语Transformer模型fr_core_news_trf(自带指代消解功能),注意法语形容词常后置,需适配依存关系标签。
步骤与代码示例
import spacy # 加载带指代消解的法语Transformer模型 nlp = spacy.load("fr_core_news_trf") def extract_noun_adj_pairs_fr(doc): pairs = set() # 处理指代链 for coref_cluster in doc.spans.get("coref", []): main_noun_span = coref_cluster[0] for ref_span in coref_cluster[1:]: ref_span._.coref_resolved = main_noun_span.text # 遍历名词短语 for noun_chunk in doc.noun_chunks: noun_root = noun_chunk.root # 法语中形容词常后置,需匹配amod和nmod等依存标签 adj_modifiers = [tok.text for tok in noun_root.children if tok.pos_ == "ADJ" and tok.dep_ in ("amod", "nmod")] # 添加当前名词-形容词对 for adj in adj_modifiers: pairs.add((noun_chunk.text, adj)) # 处理指代关系 if hasattr(noun_root, "_") and noun_root._.coref_resolved: resolved_noun = noun_root._.coref_resolved resolved_span = doc.char_span(doc.text.find(resolved_noun), doc.text.find(resolved_noun) + len(resolved_noun)) if resolved_span: resolved_adj = [tok.text for tok in resolved_span.root.children if tok.pos_ == "ADJ" and tok.dep_ in ("amod", "nmod")] for adj in resolved_adj: pairs.add((resolved_noun, adj)) # 处理连词(如et/mais)连接的并列名词 for conj_noun in noun_root.conjuncts: if conj_noun.pos_ != "NOUN": continue conj_span = doc.char_span(doc.text.find(conj_noun.text), doc.text.find(conj_noun.text) + len(conj_noun.text)) if not conj_span: continue conj_adj = [tok.text for tok in conj_span.root.children if tok.pos_ == "ADJ" and tok.dep_ in ("amod", "nmod")] # 共享原名词的形容词 for adj in adj_modifiers: pairs.add((conj_noun.text, adj)) # 添加并列名词自身的形容词对 for adj in conj_adj: pairs.add((conj_noun.text, adj)) return list(pairs) # 测试 text_fr = "Le chat grand et rouge est mignon. Il est à côté du chien petit, qui est paresseux." doc_fr = nlp(text_fr) print(extract_noun_adj_pairs_fr(doc_fr)) # 输出示例:[('Le chat grand et rouge', 'grand'), ('Le chat grand et rouge', 'rouge'), ('Le chat grand et rouge', 'mignon'), ('du chien petit', 'petit'), ('du chien petit', 'paresseux')]
关键说明
- 法语形容词常后置,需同时匹配
amod和nmod等依存标签,覆盖更多修饰场景。 - 法语指代消解的逻辑与英语完全一致,依赖模型自带的coref簇。
- 连词处理针对法语常用连词
et/mais等,通过conjuncts属性获取并列名词。
内容的提问来源于stack exchange,提问作者michel gold
相关产品推荐
相关产品推荐

