You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用蚁群优化(ACO)进行降雨数据集特征选择时遇维度错误,求解决方案

蚁群优化(ACO)特征选择代码错误解决

我尝试使用蚁群优化(Ant colony optimization, ACO)算法对降雨数据集进行特征选择,实现代码如下:

import numpy as np
from sklearn.datasets import load_breast_cancer
from sklearn.model_selection import train_test_split
from sklearn.metrics import accuracy_score
from sklearn.neighbors import KNeighborsClassifier


X = x
y = df_cap['PRECTOTCORR_SUM'] 

# Split data into training and test sets
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

# Define ACO feature selection function
def aco_feature_selection(X_train, X_test, y_train, y_test, num_ants=10, max_iter=50, alpha=1.0, beta=2.0, evaporation=0.5, q0=0.9):
    num_features = X_train.shape[1]
    pheromone = np.ones(num_features)
    best_solution = None
    best_accuracy = 0.0
    
    # Run ACO algorithm
    for i in range(max_iter):
        ant_solutions = []
        ant_accuracies = []
        
        # Generate ant solutions
        for ant in range(num_ants):
            features = np.random.choice([0, 1], size=num_features, p=[1-pheromone,pheromone])
            X_train_selected = X_train[:, features == 1]
            X_test_selected = X_test[:, features == 1]
            knn = KNeighborsClassifier()
            knn.fit(X_train_selected, y_train)
            y_pred = knn.predict(X_test_selected)
            accuracy = accuracy_score(y_test, y_pred)
            ant_solutions.append(features)
            ant_accuracies.append(accuracy)
            
            # Update best solution
            if accuracy > best_accuracy:
                best_solution = features
                best_accuracy = accuracy
        
        # Update pheromone levels
        pheromone *= evaporation
        for ant in range(num_ants):
            features = ant_solutions[ant]
            accuracy = ant_accuracies[ant]
            if accuracy >= np.mean(ant_accuracies):
                pheromone[features == 1] += alpha
            else:
                pheromone[features == 1] += beta
        
        # Apply elitism
        if best_solution is not None:
            pheromone[best_solution == 1] += q0
        
    return best_solution

# Run ACO feature selection
selected_features = aco_feature_selection(X_train, X_test, y_train, y_test)

# Print selected features
print("Selected features:", np.where(selected_features == 1)[0])

运行时出现错误:

ValueError                             
Input In [175], in aco_feature_selection(X_train, X_test, y_train, y_test, num_ants, max_iter, alpha, beta, evaporation, q0)
     26 # Generate ant solutions
     27 for ant in range(num_ants):
---> 28     features = np.random.choice([0, 1], size=num_features, p=[1-pheromone,pheromone])
     29     X_train_selected = X_train[:, features == 1]
     30     X_test_selected = X_test[:, features == 1]

File mtrand.pyx:930, in numpy.random.mtrand.RandomState.choice()

ValueError: 'p' must be 1-dimensional

尝试用flatten()处理后又出现新错误:

ValueError: 'a' and 'p' must have same size

错误原因分析

  • np.random.choice参数用法错误:你试图一次性生成num_features个特征的选择结果,但p参数传入的是长度为num_features的数组组成的列表([1-pheromone,pheromone]),这会导致p变成二维数组,不符合np.random.choice要求的一维概率分布格式。
  • 信息素未做概率映射:原代码中信息素pheromone的值会随着迭代不断累加,最终会超出[0,1]的概率范围,直接作为概率使用会导致逻辑错误。

修正后的代码

import numpy as np
from sklearn.model_selection import train_test_split
from sklearn.metrics import accuracy_score
from sklearn.neighbors import KNeighborsClassifier


X = x  # 确保x是你的特征数据集
y = df_cap['PRECTOTCORR_SUM'] 

# Split data into training and test sets
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

# Define ACO feature selection function
def aco_feature_selection(X_train, X_test, y_train, y_test, num_ants=10, max_iter=50, alpha=1.0, beta=2.0, evaporation=0.5, q0=0.9):
    num_features = X_train.shape[1]
    pheromone = np.ones(num_features)
    best_solution = None
    best_accuracy = 0.0
    
    # Run ACO algorithm
    for i in range(max_iter):
        ant_solutions = []
        ant_accuracies = []
        
        # Generate ant solutions
        for ant in range(num_ants):
            # 将信息素通过sigmoid映射到[0,1]区间,作为特征被选中的概率
            prob_select = 1 / (1 + np.exp(-pheromone))
            # 用二项分布生成特征选择向量,每个特征独立采样
            features = np.random.binomial(1, prob_select)
            
            # 处理全0特征的情况,避免KNN训练报错
            if np.sum(features) == 0:
                features[np.random.randint(num_features)] = 1
                
            X_train_selected = X_train[:, features == 1]
            X_test_selected = X_test[:, features == 1]
            knn = KNeighborsClassifier()
            knn.fit(X_train_selected, y_train)
            y_pred = knn.predict(X_test_selected)
            accuracy = accuracy_score(y_test, y_pred)
            ant_solutions.append(features)
            ant_accuracies.append(accuracy)
            
            # 更新最优解,使用深拷贝避免后续修改影响存储的最优值
            if accuracy > best_accuracy:
                best_solution = features.copy()
                best_accuracy = accuracy
        
        # 更新信息素水平
        pheromone *= evaporation
        mean_accuracy = np.mean(ant_accuracies)
        for ant in range(num_ants):
            features = ant_solutions[ant]
            accuracy = ant_accuracies[ant]
            mask = features == 1
            # 用准确率差异加权更新信息素,避免无限制累积
            if accuracy >= mean_accuracy:
                pheromone[mask] += alpha * (accuracy - mean_accuracy)
            else:
                pheromone[mask] += beta * (mean_accuracy - accuracy)
        
        # 精英策略:给最优解对应的特征额外增加信息素
        if best_solution is not None:
            pheromone[best_solution == 1] += q0 * best_accuracy
        
    return best_solution

# 执行ACO特征选择
selected_features = aco_feature_selection(X_train, X_test, y_train, y_test)

# 输出选中的特征索引
print("Selected features:", np.where(selected_features == 1)[0])

关键修正点

  • 特征采样逻辑:改用np.random.binomial对每个特征独立采样,通过sigmoid函数将信息素映射到合法的概率区间[0,1],解决原代码中概率维度不匹配的问题。
  • 边界情况处理:增加全0特征的判断与修正,避免因无特征输入导致KNN模型训练报错。
  • 信息素更新优化:用准确率差异对信息素更新量加权,防止信息素无限制累积,同时保留精英策略的正向激励效果。
  • 最优解存储优化:使用copy()深拷贝最优解,避免后续数组修改覆盖已存储的最优结果。

内容的提问来源于stack exchange,提问作者Safee987

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.31 05:27:25