使用蚁群优化(ACO)进行降雨数据集特征选择时遇维度错误,求解决方案
蚁群优化(ACO)特征选择代码错误解决
我尝试使用蚁群优化(Ant colony optimization, ACO)算法对降雨数据集进行特征选择,实现代码如下:
import numpy as np from sklearn.datasets import load_breast_cancer from sklearn.model_selection import train_test_split from sklearn.metrics import accuracy_score from sklearn.neighbors import KNeighborsClassifier X = x y = df_cap['PRECTOTCORR_SUM'] # Split data into training and test sets X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42) # Define ACO feature selection function def aco_feature_selection(X_train, X_test, y_train, y_test, num_ants=10, max_iter=50, alpha=1.0, beta=2.0, evaporation=0.5, q0=0.9): num_features = X_train.shape[1] pheromone = np.ones(num_features) best_solution = None best_accuracy = 0.0 # Run ACO algorithm for i in range(max_iter): ant_solutions = [] ant_accuracies = [] # Generate ant solutions for ant in range(num_ants): features = np.random.choice([0, 1], size=num_features, p=[1-pheromone,pheromone]) X_train_selected = X_train[:, features == 1] X_test_selected = X_test[:, features == 1] knn = KNeighborsClassifier() knn.fit(X_train_selected, y_train) y_pred = knn.predict(X_test_selected) accuracy = accuracy_score(y_test, y_pred) ant_solutions.append(features) ant_accuracies.append(accuracy) # Update best solution if accuracy > best_accuracy: best_solution = features best_accuracy = accuracy # Update pheromone levels pheromone *= evaporation for ant in range(num_ants): features = ant_solutions[ant] accuracy = ant_accuracies[ant] if accuracy >= np.mean(ant_accuracies): pheromone[features == 1] += alpha else: pheromone[features == 1] += beta # Apply elitism if best_solution is not None: pheromone[best_solution == 1] += q0 return best_solution # Run ACO feature selection selected_features = aco_feature_selection(X_train, X_test, y_train, y_test) # Print selected features print("Selected features:", np.where(selected_features == 1)[0])
运行时出现错误:
ValueError Input In [175], in aco_feature_selection(X_train, X_test, y_train, y_test, num_ants, max_iter, alpha, beta, evaporation, q0) 26 # Generate ant solutions 27 for ant in range(num_ants): ---> 28 features = np.random.choice([0, 1], size=num_features, p=[1-pheromone,pheromone]) 29 X_train_selected = X_train[:, features == 1] 30 X_test_selected = X_test[:, features == 1] File mtrand.pyx:930, in numpy.random.mtrand.RandomState.choice() ValueError: 'p' must be 1-dimensional
尝试用flatten()处理后又出现新错误:
ValueError: 'a' and 'p' must have same size
错误原因分析
np.random.choice参数用法错误:你试图一次性生成num_features个特征的选择结果,但p参数传入的是长度为num_features的数组组成的列表([1-pheromone,pheromone]),这会导致p变成二维数组,不符合np.random.choice要求的一维概率分布格式。- 信息素未做概率映射:原代码中信息素
pheromone的值会随着迭代不断累加,最终会超出[0,1]的概率范围,直接作为概率使用会导致逻辑错误。
修正后的代码
import numpy as np from sklearn.model_selection import train_test_split from sklearn.metrics import accuracy_score from sklearn.neighbors import KNeighborsClassifier X = x # 确保x是你的特征数据集 y = df_cap['PRECTOTCORR_SUM'] # Split data into training and test sets X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42) # Define ACO feature selection function def aco_feature_selection(X_train, X_test, y_train, y_test, num_ants=10, max_iter=50, alpha=1.0, beta=2.0, evaporation=0.5, q0=0.9): num_features = X_train.shape[1] pheromone = np.ones(num_features) best_solution = None best_accuracy = 0.0 # Run ACO algorithm for i in range(max_iter): ant_solutions = [] ant_accuracies = [] # Generate ant solutions for ant in range(num_ants): # 将信息素通过sigmoid映射到[0,1]区间,作为特征被选中的概率 prob_select = 1 / (1 + np.exp(-pheromone)) # 用二项分布生成特征选择向量,每个特征独立采样 features = np.random.binomial(1, prob_select) # 处理全0特征的情况,避免KNN训练报错 if np.sum(features) == 0: features[np.random.randint(num_features)] = 1 X_train_selected = X_train[:, features == 1] X_test_selected = X_test[:, features == 1] knn = KNeighborsClassifier() knn.fit(X_train_selected, y_train) y_pred = knn.predict(X_test_selected) accuracy = accuracy_score(y_test, y_pred) ant_solutions.append(features) ant_accuracies.append(accuracy) # 更新最优解,使用深拷贝避免后续修改影响存储的最优值 if accuracy > best_accuracy: best_solution = features.copy() best_accuracy = accuracy # 更新信息素水平 pheromone *= evaporation mean_accuracy = np.mean(ant_accuracies) for ant in range(num_ants): features = ant_solutions[ant] accuracy = ant_accuracies[ant] mask = features == 1 # 用准确率差异加权更新信息素,避免无限制累积 if accuracy >= mean_accuracy: pheromone[mask] += alpha * (accuracy - mean_accuracy) else: pheromone[mask] += beta * (mean_accuracy - accuracy) # 精英策略:给最优解对应的特征额外增加信息素 if best_solution is not None: pheromone[best_solution == 1] += q0 * best_accuracy return best_solution # 执行ACO特征选择 selected_features = aco_feature_selection(X_train, X_test, y_train, y_test) # 输出选中的特征索引 print("Selected features:", np.where(selected_features == 1)[0])
关键修正点
- 特征采样逻辑:改用
np.random.binomial对每个特征独立采样,通过sigmoid函数将信息素映射到合法的概率区间[0,1],解决原代码中概率维度不匹配的问题。 - 边界情况处理:增加全0特征的判断与修正,避免因无特征输入导致KNN模型训练报错。
- 信息素更新优化:用准确率差异对信息素更新量加权,防止信息素无限制累积,同时保留精英策略的正向激励效果。
- 最优解存储优化:使用
copy()深拷贝最优解,避免后续数组修改覆盖已存储的最优结果。
内容的提问来源于stack exchange,提问作者Safee987
相关产品推荐
相关产品推荐

