基于Iris数据集实现多分类Logistic Regression遇维度不匹配错误求助
修复多分类Logistic Regression的维度不匹配错误
问题场景
在Kaggle的Iris数据集上手动实现多分类Logistic Regression(一对多策略),运行代码时出现维度不匹配的ValueError,同时需要正确生成one-hot格式的分类标签。
原实现代码
import numpy as np import pandas as pd from sklearn.model_selection import train_test_split def standardize(X_tr): # (x-Mean(x))/std(X) Normalizes data for i in range(X_tr.shape[1]): X_tr[:, i] = (X_tr[:, i] - np.mean(X_tr[:, i])) / np.std(X_tr[:, i]) return X_tr def sigmoid(z): #Sigmoid/Logistic function sig = 1 / (1 + np.exp(-z)) return sig def cost(theta, X, y): z = np.dot(X, theta) cost0 = y.T.dot(np.log(sigmoid(z))) cost1 = (1 - y).T.dot(np.log(1 - sigmoid(z))) cost = -((cost1 + cost0)) / len(y) return cost def initialize(X): #Initializing X feature matrix and Theta vector thetas = np.zeros((X.shape[1] + 1, len(np.unique(y)))) X = np.c_[np.ones((X.shape[0], 1)), X] # adding 691 rows of ones as the first column in X return thetas, X def fit(X, y, alpha=0.01, iterations=1000): # Gradient Descent thetas_list = [] X = np.c_[np.ones((X.shape[0], 1)), X] for i in range(len(np.unique(y))): y_one_vs_all = np.where(y == np.unique(y)[i], 1, 0) thetas, _ = initialize(X) for j in range(iterations): z = np.dot(X, thetas[:, i]) h = sigmoid(z) gradient = np.dot(X.T, (h - y_one_vs_all)) / len(y) thetas[:, i] -= alpha * gradient thetas_list.append(thetas[:, i]) global gthetas gthetas = thetas_list return None def predict(X): X = np.c_[np.ones((X.shape[0], 1)), X] predictions = [] for sample in X: probs = [] for thetas in gthetas: z = np.dot(sample, thetas) probs.append(sigmoid(z)) predictions.append(np.argmax(probs) + 1) return predictions # load data df = pd.read_csv("Iris.csv") X = df.iloc[:, :-1].values y = df.iloc[:, -1].values # convert class categorical values to numerical values df['Species'].replace('Iris-setosa', 1, inplace=True) df['Species'].replace('Iris-versicolor', 2, inplace=True) df['Species'].replace('Iris-virginica', 3, inplace=True) # prepare one-vs-all labels for multiclass classification y1 = pd.DataFrame(np.zeros((len(y), len(np.unique(y))))) for i in range(len(np.unique(y))): for j in range(len(y1)): if y[j] == np.unique(y)[i]: y1.iloc[j, i] = 1 else: y1.iloc[j, i] = 0 # split data into training and test sets X_train, X_test, y_train, y_test = train_test_split(X, y1, test_size=0.2, random_state=0) # standardize features X_train = standardize(X_train) X_test = standardize(X_test) # fit logistic regression model fit(X_train, y_train, alpha=0.01, iterations=400) # make predictions on test set predictions = predict(X_test) print(predictions)
运行错误信息
ValueError Traceback (most recent call last) ~\AppData\Local\Temp/ipykernel_14368/3997569506.py in <module> 1 # standardize features 2 X_train = standardize(X_train) ----> 3 thetas_list = fit(X_train, y_train) 4 plt.scatter(range(len(cost_list)), cost_list, c="blue") 5 plt.show() ~\AppData\Local\Temp/ipykernel_14368/3827160719.py in fit(X, y, alpha, iter) 6 thetas, _ = initialize(X) 7 for j in range(iter): ----> 8 z = dot(X, thetas[:, i]) 9 h = sigmoid(z) 10 gradient = dot(X.T, (h - y_one_vs_all)) / len(y) <__array_function__ internals> in dot(*args, **kwargs) ValueError: shapes (120,6) and (7,) not aligned: 6 (dim 1) != 7 (dim 0)
问题分析
- 重复添加偏置列导致维度不匹配:
fit函数中已经给X添加了偏置列,但initialize函数再次给X添加偏置列,导致X维度异常,初始化的theta维度与X不匹配,最终矩阵点乘失败。 - 标签处理逻辑错误:
fit函数传入的y_train是one-hot格式,但y_one_vs_all的生成逻辑基于one-hot数组的唯一值,无法正确生成一对多的二分类标签。 - 全局变量使用不规范:依赖全局变量传递模型参数,代码耦合性高且易出错。
- one-hot标签生成效率低:嵌套循环生成标签,代码冗余且运行效率差。
修复步骤
1. 修正initialize函数
移除对X的偏置列操作,仅初始化对应每个类别的theta向量:
def initialize(num_features): return np.zeros(num_features)
2. 重构fit函数
- 移除全局变量,改为返回训练好的theta列表和类别映射
- 传入原始类别数值标签,而非one-hot标签
- 针对每个类别单独训练一对多分类器:
def fit(X, y, alpha=0.01, iterations=1000): thetas_list = [] X = np.c_[np.ones((X.shape[0], 1)), X] unique_classes = np.unique(y) for cls in unique_classes: y_one_vs_all = np.where(y == cls, 1, 0) theta = initialize(X.shape[1]) for _ in range(iterations): z = np.dot(X, theta) h = sigmoid(z) gradient = np.dot(X.T, (h - y_one_vs_all)) / len(y) theta -= alpha * gradient thetas_list.append(theta) return thetas_list, unique_classes
3. 优化one-hot标签生成
用numpy的eye函数替代嵌套循环,高效生成标签:
y_num = df['Species'].values y_onehot = np.eye(len(np.unique(y_num)))[y_num - 1] # 类别1/2/3转为0/1/2索引
4. 修正predict函数
接收训练好的参数,避免全局变量依赖:
def predict(X, thetas_list, unique_classes): X = np.c_[np.ones((X.shape[0], 1)), X] predictions = [] for sample in X: probs = [sigmoid(np.dot(sample, theta)) for theta in thetas_list] pred_idx = np.argmax(probs) predictions.append(unique_classes[pred_idx]) return predictions
完整修正代码
import numpy as np import pandas as pd from sklearn.model_selection import train_test_split def standardize(X_tr): for i in range(X_tr.shape[1]): X_tr[:, i] = (X_tr[:, i] - np.mean(X_tr[:, i])) / np.std(X_tr[:, i]) return X_tr def sigmoid(z): return 1 / (1 + np.exp(-z)) def cost(theta, X, y): z = np.dot(X, theta) h = sigmoid(z) # 避免log(0)的数值错误,添加极小值 cost = -np.mean(y * np.log(h + 1e-10) + (1 - y) * np.log(1 - h + 1e-10)) return cost def initialize(num_features): return np.zeros(num_features) def fit(X, y, alpha=0.01, iterations=1000): thetas_list = [] X = np.c_[np.ones((X.shape[0], 1)), X] unique_classes = np.unique(y) for cls in unique_classes: y_one_vs_all = np.where(y == cls, 1, 0) theta = initialize(X.shape[1]) for _ in range(iterations): z = np.dot(X, theta) h = sigmoid(z) gradient = np.dot(X.T, (h - y_one_vs_all)) / len(y) theta -= alpha * gradient thetas_list.append(theta) return thetas_list, unique_classes def predict(X, thetas_list, unique_classes): X = np.c_[np.ones((X.shape[0], 1)), X] predictions = [] for sample in X: probs = [sigmoid(np.dot(sample, theta)) for theta in thetas_list] pred_idx = np.argmax(probs) predictions.append(unique_classes[pred_idx]) return predictions # 加载数据 df = pd.read_csv("Iris.csv") X = df.iloc[:, :-1].values # 转换类别为数值 df['Species'] = df['Species'].map({ 'Iris-setosa': 1, 'Iris-versicolor': 2, 'Iris-virginica': 3 }) y = df['Species'].values # 拆分数据集 X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=0) # 标准化特征 X_train = standardize(X_train) X_test = standardize(X_test) # 训练模型 thetas_list, unique_classes = fit(X_train, y_train, alpha=0.01, iterations=1000) # 预测测试集 predictions = predict(X_test, thetas_list, unique_classes) print("测试集预测结果:", predictions) # 计算准确率 accuracy = np.mean(predictions == y_test) print("测试集准确率:", accuracy)
额外说明
- 修正后的代码移除了全局变量,参数传递更清晰
- 优化了损失函数计算,避免
log(0)引发的数值错误 - 简化了one-hot标签生成逻辑
- 新增准确率计算,方便评估模型性能
内容的提问来源于stack exchange,提问作者Ashar
相关产品推荐
相关产品推荐

