You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于KMeans聚类的ODKM离群检测算法柱状图与直方图绘制优化问题

基于KMeans的离群检测算法可视化问题解决方案

问题描述

在为基于KMeans的聚类算法绘制柱状图时,需要实现离群簇位于x轴末端、其余簇相互紧邻的展示效果。默认x轴刻度为等距分布:

---|---|---|-----------------> x-axis
0  1   2   3 

需要调整分箱宽度,实现非等距x轴排列,让基于Score预测得到的标签为3的离群簇与其余簇拉开距离:

---|---|--------------|------> x-axis
0  1   2              3 

解决方案

修改后的ODKM类代码(新增簇标签输出能力)

from sklearn.cluster import KMeans
import seaborn as sns
import numpy as np
from pandas import DataFrame
from math import pow
import math

class ODKM:
    
    def __init__(self,n_clusters=15,effectiveness=500,max_iter=2, random_state=42):
        self.n_clusters=n_clusters
        self.effectiveness=effectiveness
        self.max_iter=max_iter
        self.random_state = random_state
        self.kmeans = {}
        self.cluster_score = {}
        # 存储各特征的簇映射关系:按簇中心升序排列后,原始标签映射为新的顺序标签,保证离群簇标签值最大
        self.cluster_map = {}
        
    def fit(self, data):
        length = len(data)
        for column in data.columns:
            kmeans = KMeans(n_clusters=self.n_clusters,max_iter=self.max_iter, random_state=self.random_state)
            self.kmeans[column]=kmeans
            kmeans.fit(data[column].values.reshape(-1,1))
            assign = DataFrame(kmeans.predict(data[column].values.reshape(-1,1)),columns=['cluster'])
            cluster_score=assign.groupby('cluster').apply(len).apply(lambda x:x/length)
            ratio=cluster_score.copy()
        
            sorted_centers = sorted(kmeans.cluster_centers_)
            max_distance = ( sorted_centers[-1] - sorted_centers[0] )[ 0 ]
            # 生成原始标签到升序标签的映射
            cluster_idx_map = {tuple(v)[0]:i for i,v in enumerate(sorted_centers)}
            self.cluster_map[column] = {k:cluster_idx_map[tuple(v)[0]] for k,v in enumerate(kmeans.cluster_centers_)}
        
            for i in range(self.n_clusters):
                for k in range(self.n_clusters):
                    if i != k:
                        dist = abs(kmeans.cluster_centers_[i] - kmeans.cluster_centers_[k])/max_distance
                        effect = ratio[k]*(1/pow(self.effectiveness,dist))
                        cluster_score[i] = cluster_score[i]+effect
                        
            self.cluster_score[column] = cluster_score
                    
    def predict(self, data):
        length = len(data)
        score_array = np.zeros(length)
        # 存储全局簇标签:取所有特征维度的最大标签作为样本最终簇标签,离群簇优先级最高
        global_label = np.zeros(length, dtype=int)
        for column in data.columns:
            kmeans = self.kmeans[ column ]
            cluster_score = self.cluster_score[ column ]
            c_map = self.cluster_map[column]
            assign = kmeans.predict( data[ column ].values.reshape(-1,1) )
            # 按映射转换为升序标签
            mapped_assign = np.array([c_map[x] for x in assign])
            global_label = np.max([global_label, mapped_assign], axis=0)
            
            for i in range(length):
                score_array[i] = score_array[i] + math.log10( cluster_score[assign[i]] )
            
        return score_array, global_label
    
    def fit_predict(self,data):
        self.fit(data)
        return self.predict(data)

测试与数据预处理

import pandas as pd
import matplotlib.pyplot as plt

df = pd.DataFrame(data={'attr1':[1,1,1,1,2,2,2,2,2,2,2,2,3,5,5,6,6,7,7,7,7,7,7,7,15],
                        'attr2':[1,1,1,1,2,2,2,2,2,2,2,2,3,5,5,6,6,7,7,7,13,13,13,14,15]})

odkm_model = ODKM(n_clusters=3, max_iter=1)
# 同时获取异常分数和簇标签
score, cluster_label = odkm_model.fit_predict(df)

df['ODKM_Score'] = score 
df['Cluster_label'] = cluster_label
# 和现有逻辑对齐取绝对值
df['Score'] = df['ODKM_Score'].abs()

可视化实现

1. 非等距x轴柱状图

# 统计每个簇的样本数量
cluster_cnt = df['Cluster_label'].value_counts().sort_index()
# 自定义x轴位置:前两个簇紧邻,第三个簇偏移2个单位拉开距离
x_pos = [0, 1, 3]
colors = ["#00f0f0","#ff0000","#00ff00"]

plt.figure(figsize=(8,4))
plt.bar(x_pos, cluster_cnt.values, color=colors, width=0.8)
# 替换x轴刻度为簇标签
plt.xticks(x_pos, cluster_cnt.index)
plt.xlabel('Cluster Label')
plt.ylabel('Sample Count')
plt.title('Cluster Distribution with Outlier Isolation')
plt.show()

2. 按簇配色的直方图+KDE曲线

plt.figure(figsize=(8,4))
# 手动指定分箱边界,保证离群簇分箱和普通簇分箱拉开距离,解决x轴间距不一致问题
bins = [0.4, 0.6, 1.1, 2.1]
sns.histplot(data=df, x='Score', hue='Cluster_label', palette=colors, alpha=1, bins=bins, edgecolor='white')
# 新增双Y轴绘制KDE曲线
ax2 = plt.gca().twinx()
sns.kdeplot(data=df, x='Score', hue='Cluster_label', palette=colors, ax=ax2, lw=2, legend=False)
plt.title('Score Distribution with Cluster Color Coding')
plt.show()

内容的提问来源于stack exchange,提问作者Mario

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.30 12:15:05