如何用Python从5万+特征中选2个特征可视化区分Sarcopenia样本
问题描述
我有一个CSV格式的数据库synthetic_feature_file,包含5万余个已处理的非原始特征,共43个样本。我希望从中找到两个特征,能将Sarcopenia样本与非Sarcopenia样本区分开,并用这两个特征作为坐标轴绘制散点图,使两类样本形成无重叠的聚类。目前通过随机选取特征反复运行代码查看结果,效率极低,请问该如何修改现有代码?
现有代码:
import numpy as np import pandas as pd import matplotlib.pyplot as plt # Read synthesized feature data syn_data = pd.read_csv(synthetic_feature_file) # labeled sample(people who suffer from Sarcopenia) sample_indices = [1, 6, 7, 11, 14, 15, 27] # Randomly pick two features as x and y axes x_feature = np.random.choice(syn_data.columns[0:10000]) y_feature = np.random.choice(syn_data.columns[10001:20000]) # Clean feature names and remove illegal characters x_feature = x_feature.strip().replace('\t', '') y_feature = y_feature.strip().replace('\t', '') plt.figure(figsize=(8, 6)) # Other samples(people who did not suffer from Sarcopenia) other_samples = syn_data.drop(sample_indices) plt.scatter(other_samples[x_feature], other_samples[y_feature], color='blue', label='Other Samples') # Red sample red_samples = syn_data.iloc[sample_indices] plt.scatter(red_samples[x_feature], red_samples[y_feature], color='red', label='Sample Indices') plt.xlabel(x_feature) plt.ylabel(y_feature) plt.title("Visualization") plt.legend() plt.show()
解决方案
核心思路是先量化单个特征的分类区分能力,筛选出对两类样本区分度高的特征,再在这些高区分度特征中两两组合验证,自动找出能让两类样本无重叠的特征对,最后绘图,彻底替代随机尝试的低效方式。
步骤说明
- 计算特征区分度:用Cohen's d指标(两类样本均值差除以合并标准差)衡量单个特征的区分能力,值越大说明特征越能分开两类样本
- 筛选高区分度特征:设定阈值(比如d>1)缩小候选范围,避免遍历5万特征的海量组合
- 验证特征对无重叠:遍历筛选后的特征对,检查两类样本在二维空间中是否完全无重叠,找到符合条件的组合后自动绘图
修改后的代码
import numpy as np import pandas as pd import matplotlib.pyplot as plt # 读取数据 syn_data = pd.read_csv("synthetic_feature_file") # 标记Sarcopenia样本索引 sarc_indices = [1, 6, 7, 11, 14, 15, 27] # 拆分两类样本 sarc_samples = syn_data.iloc[sarc_indices].reset_index(drop=True) non_sarc_samples = syn_data.drop(sarc_indices).reset_index(drop=True) # 计算单个特征的Cohen's d(区分度指标) def calculate_cohens_d(feature): sarc_vals = sarc_samples[feature].values non_sarc_vals = non_sarc_samples[feature].values mean_diff = np.mean(sarc_vals) - np.mean(non_sarc_vals) pooled_std = np.sqrt(((len(sarc_vals)-1)*np.var(sarc_vals) + (len(non_sarc_vals)-1)*np.var(non_sarc_vals)) / (len(sarc_vals)+len(non_sarc_vals)-2)) if pooled_std == 0: return 0 # 标准差为0说明该特征无区分度 return abs(mean_diff / pooled_std) # 计算所有特征的区分度,筛选高区分度特征(这里取d>1,可根据需求调整) feature_scores = {col: calculate_cohens_d(col) for col in syn_data.columns} high_d_features = [col for col, score in feature_scores.items() if score > 1] print(f"筛选出高区分度特征数量:{len(high_d_features)}") # 检查特征对是否能让两类样本无重叠 def check_no_overlap(x_col, y_col): # 获取两类样本的x、y值范围 sarc_x_min, sarc_x_max = sarc_samples[x_col].min(), sarc_samples[x_col].max() sarc_y_min, sarc_y_max = sarc_samples[y_col].min(), sarc_samples[y_col].max() non_sarc_x_min, non_sarc_x_max = non_sarc_samples[x_col].min(), non_sarc_samples[x_col].max() non_sarc_y_min, non_sarc_y_max = non_sarc_samples[y_col].min(), non_sarc_samples[y_col].max() # 判断是否在x轴或y轴上完全分离,或二维矩形无交集 if (sarc_x_max < non_sarc_x_min or sarc_x_min > non_sarc_x_max) or (sarc_y_max < non_sarc_y_min or sarc_y_min > non_sarc_y_max): return True return False # 遍历寻找符合条件的特征对 valid_pairs = [] for i in range(len(high_d_features)): for j in range(i+1, len(high_d_features)): x_col = high_d_features[i] y_col = high_d_features[j] if check_no_overlap(x_col, y_col): valid_pairs.append((x_col, y_col)) # 如果只需要一对符合条件的特征,找到后可直接break退出循环 # break # if valid_pairs: # break print(f"找到符合条件的特征对数量:{len(valid_pairs)}") # 绘制符合条件的特征对散点图(最多绘制前5个,可调整) if valid_pairs: for idx, (x_feature, y_feature) in enumerate(valid_pairs[:5]): plt.figure(figsize=(8, 6)) plt.scatter(non_sarc_samples[x_feature], non_sarc_samples[y_feature], color='blue', label='非Sarcopenia样本') plt.scatter(sarc_samples[x_feature], sarc_samples[y_feature], color='red', label='Sarcopenia样本') plt.xlabel(x_feature) plt.ylabel(y_feature) plt.title(f"无重叠聚类可视化 - 特征对{idx+1}") plt.legend() plt.show() else: print("未找到能使两类样本无重叠的特征对,建议降低区分度阈值或调整无重叠判断条件")
关键优化点
- 效率提升:先筛选高区分度特征,把5万特征的组合量压缩到数千甚至数百次,避免无意义的随机尝试
- 可定制性:可以根据需求调整区分度阈值(比如把d>1改成d>0.8)或无重叠判断逻辑(比如允许少量重叠)
- 自动化:无需手动干预,代码会自动找到符合条件的特征对并绘图
内容的提问来源于stack exchange,提问作者董珈妤
相关产品推荐
相关产品推荐

