Python中如何删除DataFrame逗号分隔字符串的None并统计行内高频词汇
实现代码
import pandas as pd from collections import Counter # 构造示例数据,实际使用时替换为你自己的DataFrame df = pd.DataFrame({ 'Col': [ 'Car, None, None, Car, Bus, None', 'None', 'Bus, Bus, None, Car, Car, None', 'None, None, None' ] }) # 处理每行内容:拆分、过滤None、标记全None行 def parse_row(raw_str): word_list = [word.strip() for word in raw_str.split(',') if word.strip() != 'None'] return word_list if word_list else None df['temp_word_list'] = df['Col'].apply(parse_row) # 删除全None行 df = df.dropna(subset=['temp_word_list']).reset_index(drop=True) # 生成处理后的Col列 df['Col'] = df['temp_word_list'].apply(lambda x: ', '.join(x)) # 统计每行最高频词汇,支持并列最高 def get_top_count(word_list): count = Counter(word_list) max_count = max(count.values()) top_list = [f"{word} ({num})" for word, num in count.items() if num == max_count] return ', '.join(top_list) df['Most common words'] = df['temp_word_list'].apply(get_top_count) # 移除临时辅助列 df = df.drop(columns=['temp_word_list']) # 输出结果 print(df)
运行上述代码得到的输出完全匹配预期示例。
内容的提问来源于stack exchange,提问作者PSCM
相关产品推荐
相关产品推荐

