Pandas DataFrame中按in/tar列标记合并title列对应位置元组的实现问题
调整后实现代码
import pandas as pd import spacy # 按需替换为你使用的spaCy模型 nlp = spacy.load("en_core_web_sm") # 直接指定待处理列,避免因列顺序变化导致索引错位 cols = ["in", "tar"] def process_single_row(row): # 生成基础依存句法元组列表,过滤标点 doc = nlp(row["title"]) dep_list = [(token.text, token.pos_, token.dep_) for token in doc if token.pos_ != "PUNCT"] list_len = len(dep_list) # 预标记每个位置所属的分组 pos_belong = [None] * list_len for col in cols: for pos in row[col]: pos_belong[pos] = col new_dep_list = [] idx = 0 while idx < list_len: # 不属于任何分组的元素直接保留 if pos_belong[idx] is None: new_dep_list.append(dep_list[idx]) idx += 1 continue # 属于指定分组的元素统一合并 current_group = pos_belong[idx] # 收集分组内所有位置的单词 merge_word_list = [dep_list[pos][0] for pos in row[current_group]] # 生成符合规则的合并元组 merged_word = f'<{current_group.upper()}>{" ".join(merge_word_list)}</{current_group.upper()}>' merged_tuple = (merged_word, current_group.upper(), "") new_dep_list.append(merged_tuple) # 跳过所有已合并的位置 idx = max(row[current_group]) + 1 return new_dep_list # 逐行处理更新title列 df["title"] = df.apply(process_single_row, axis=1)
核心调整说明
- 新增位置预标记逻辑:提前记录每个下标对应的分组(
in/tar),避免逐元素判断时遗漏多位置合并场景 - 替换原有逐位置修改逻辑:遇到属于某分组的位置时,直接拉取该分组所有位置的单词拼接,生成符合规则的合并元组
- 自动跳过已合并位置,避免重复处理同分组元素
- 修复原始代码的语法错误、缩进问题,改用pandas
apply方法逐行处理,逻辑更清晰,稳定性更高
内容的提问来源于stack exchange,提问作者joasa
相关产品推荐
相关产品推荐

