You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

机器学习:矩阵分解处理评分NaN值时MSE返回NaN的问题

问题:处理NaN值

我曾尝试将NaN值替换为0来测试输出,但即使将NaN替换为0,MSE仍返回NaN,这让我认为负责预测的Matrix Factorization类存在问题。此外,我不应将NaN替换为0,因为这会影响评分结果。

我最初从文件中读取用户和电影数据,数据表头如下:

movie_inds  0     1     2     3     4     5     6     7     8     9     ...    
user_inds                                                               ...  
0            3.0   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN  ...  
1            NaN   3.0   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN  ...  
2            NaN   NaN   1.0   NaN   NaN   NaN   3.0   NaN   4.0   NaN  ...  
3            NaN   NaN   NaN   2.0   NaN   NaN   4.0   NaN   4.0   4.0  ...  
4            NaN   NaN   NaN   NaN   1.0   NaN   NaN   NaN   NaN   NaN  ...

movie_inds  1672  1673  1674  1675  1676  1677  1678  1679  1680  1681  
user_inds  
0            NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN  
1            NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN  
2            NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN  
3            NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN  
4            NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN   NaN  

\[5 rows x 1682 columns\]

NaN值是预期存在的,因为用户并未对所有电影评分。

随后我使用Matrix Factorization转换数组并通过预测处理NaN值:

# MatrixFactorization class
class MatrixFactorization():
    # constructor
    def __init__(self, rating_matrix, rating_matrix_val, k=5, lmbda=0.01, max_epochs=50, lr=0.1):
        '''
        rating_matrix: the matrix of rating (row: user, col: movie)
        k : int, default=2
        Number of latent features
        lmbda : float, default=0.01
        Regularization parameter
        max_epochs : int, default=15
        Max number of iterations to run
        lr: float, default:0.1
        Learning rate
        '''
        self.rating_matrix = rating_matrix
        self.rating_matrix_val = rating_matrix_val
        self.k = k
        self.lmbda = lmbda
        self.max_epochs = max_epochs
        self.lr = lr
        self.U = np.random.normal(scale=1./self.k, size=(rating_matrix.shape[0], self.k))
        self.V = np.random.normal(scale=1./self.k, size=(self.k, rating_matrix.shape[1]))
        self.b_u = np.zeros(rating_matrix.shape[0])
        self.b_i = np.zeros(rating_matrix.shape[1])
        self.b = np.mean(rating_matrix[np.where(rating_matrix != 0)])
        self.mean_squared_errors = []
        self.mean_squared_errors_val = []

    # fitting the model
    def fit(self):
        for it in range(self.max_epochs):
            for i in range(len(self.rating_matrix)):
                for j in range(len(self.rating_matrix[i])):
                    if self.rating_matrix[i][j] > 0:
                        eij = self.rating_matrix[i][j] - (self.predict_rating_user_movie(i, j))
                        self.b_u[i] += self.lr * (eij - self.lmbda * self.b_u[i])
                        self.b_i[j] += self.lr * (eij - self.lmbda * self.b_i[j])
                        self.U[i,:] += self.lr * (eij * self.V[:,j] - self.lmbda * self.U[i,:])
                        self.V[:,j] += self.lr * (eij * self.U[i,:] - self.lmbda * self.V[:,j])
    
            self.mean_squared_errors.append(self.mse_training(self.rating_matrix, self.predict()))
            self.mean_squared_errors_val.append(self.mse_validation(self.rating_matrix, self.rating_matrix_val, self.predict()))
    
    # reporting model's mse using the training set
    def mse_training(self, true_rating, pred_rating):
        '''
        pred_matrix: the predict matrix of rating (row: user, col: movie)
        return: mean squared error
        '''
        error = 0
        for rt, rp in zip(true_rating, pred_rating):
            for vt, vp in zip(rt, rp):
                if vt > 0:
                    error += pow(vt - vp, 2)
        return np.sqrt(error)
    
    # reporting model's mse using testing set
    def mse_validation(self, true_rating, test_rating, pred_rating):
        '''
        pred_matrix: the predict matrix of rating (row: user, col: movie)
        return: mean squared error
        '''
        error = 0
        for rt, rtt, rp in zip(true_rating, test_rating, pred_rating):
            for vt, vtt, vp in zip(rt, rtt, rp):
                if vt == -1:
                    error += pow(vtt - vp, 2)
        return np.sqrt(error)
    
    # predicting the user's movie rating
    def predict_rating_user_movie(self, i, j):
        '''
        i: user row
        j: movie col
        '''
        return self.b + self.b_u[i] + self.b_i[j] + np.dot(self.U[i,:],self.V[:,j])
    
    # rating prediction
    def predict(self):
        '''
        return: rating prediction
        '''
        return self.b + self.b_u[:,np.newaxis] + self.b_i[np.newaxis:, ] + np.dot(self.U, self.V)
    
    # plotting model's loss
    def plot_loss(self):
        iters = [i for i in range(self.max_epochs)]
        plt.xlabel("Iterations")
        plt.ylabel("Mean Squared Error")
        plt.plot(iters, self.mean_squared_errors)
        plt.plot(iters, self.mean_squared_errors_val)
    
    # mean square error result
    def mse_result(self):
        return self.mean_squared_errors[-1]

我遇到的问题在于以下代码部分,MSE返回NaN,且我的矩阵分解类未能正确处理NaN值:

# grid search function implementation
def grid_search(parameters):
    best_mse = 1000 # choose the maximum mse initially
    best_params = None  # initialize best parameters as None

    # loop over all parameters that are passed through the grid_search function
    for params in parameters:
        k, lmbda, max_epochs, lr = params  # unpack the parameters

        # make your model with all the parameters
        mf_model = MatrixFactorization(rating_matrix_training, rating_matrix_test, k=k, lmbda=lmbda, max_epochs=max_epochs, lr=lr)

        # fit model
        mf_model.fit()

        # predict labels
        pred_ratings = mf_model.predict()

        # measure training mse
        mse_train = mf_model.mse_training(rating_matrix_training, pred_ratings)

        # measure validation mse
        mse_val = mf_model.mse_validation(rating_matrix_training, rating_matrix_test, pred_ratings)

        # find the best parameters with the minimum validation mse and update best_params
        if mse_val < best_mse:
            best_mse = mse_val
            best_params = params

        print("Parameters:", params, "Training MSE:", mse_train, "Validation MSE:", mse_val)

    # print the minimum mse
    print("Minimum Validation MSE:", best_mse)

    # return the best parameters
    return best_params

超参数调优:

import itertools

# input your parameters
k = (5, 10, 20)
lmbda = (0.01, 0.1)
e = (20, 50)
lr = (0.001, 0.01, 0.5)

# merge all parameters in one list
parameters= [k,lmbda,e,lr]
parameters= list(itertools.product(*parameters))

# run the grid search
best_params = grid_search(parameters)

现寻求帮助排查:为何替换NaN为0后MSE仍返回NaN?自定义MatrixFactorization类为何无法正确处理NaN值?如何解决该问题以完成用户评分矩阵的预测与评估?


问题排查与解决方法

1. 替换NaN为0后MSE仍返回NaN的核心原因

  • 学习率过大导致数值爆炸:你设置的lr=0.5远超合理范围,参数更新时会出现梯度爆炸,最终预测值或参数变为无穷大,计算误差时出现inf-inf的情况,返回NaN。
  • 均值计算残留NaN:如果替换NaN为0的操作不彻底(比如未重新赋值给原矩阵),self.b = np.mean(rating_matrix[np.where(rating_matrix != 0)])会因数组包含NaN导致均值为NaN,后续所有计算都会继承NaN。
  • 参数更新逻辑错误:更新U和V时,先修改U再用修改后的U计算V的更新,导致梯度计算偏差,加剧数值不稳定。

2. MatrixFactorization类无法处理NaN的原因

  • 未明确过滤NaN:初始化均值、训练循环、MSE计算中,仅通过rating_matrix[i][j]>0判断有效评分,但未明确排除NaN,当矩阵存在NaN时,NaN>0返回False会跳过,但如果均值计算或预测过程中混入NaN,会直接导致结果异常。
  • MSE计算逻辑错误:原代码返回的是误差平方和的平方根,而非真正的均方根误差(RMSE),且未统计有效样本数,若有效样本为0或混入NaN,会直接返回NaN。

3. 具体修复步骤

步骤1:正确处理原始矩阵的NaN

无需替换NaN为0,直接在代码中过滤NaN:

# 修改__init__中的均值计算
valid_mask = ~np.isnan(self.rating_matrix) & (self.rating_matrix > 0)
valid_ratings = self.rating_matrix[valid_mask]
self.b = np.mean(valid_ratings) if len(valid_ratings) >0 else 0.0
步骤2:修复参数更新逻辑

更新U和V前保存原始值,避免互相干扰:

def fit(self):
    for it in range(self.max_epochs):
        for i in range(self.rating_matrix.shape[0]):
            for j in range(self.rating_matrix.shape[1]):
                # 明确过滤NaN和无效评分
                if not np.isnan(self.rating_matrix[i][j]) and self.rating_matrix[i][j] > 0:
                    eij = self.rating_matrix[i][j] - self.predict_rating_user_movie(i, j)
                    # 保存原始参数值
                    u_old = self.U[i,:].copy()
                    v_old = self.V[:,j].copy()
                    # 更新偏置
                    self.b_u[i] += self.lr * (eij - self.lmbda * self.b_u[i])
                    self.b_i[j] += self.lr * (eij - self.lmbda * self.b_i[j])
                    # 更新U和V
                    self.U[i,:] += self.lr * (eij * v_old - self.lmbda * u_old)
                    self.V[:,j] += self.lr * (eij * u_old - self.lmbda * v_old)
    
        pred = self.predict()
        self.mean_squared_errors.append(self.mse_training(self.rating_matrix, pred))
        self.mean_squared_errors_val.append(self.mse_validation(self.rating_matrix, self.rating_matrix_val, pred))
步骤3:修复MSE计算逻辑

计算真正的RMSE,明确过滤NaN并统计有效样本数:

def mse_training(self, true_rating, pred_rating):
    error = 0.0
    count = 0
    for rt, rp in zip(true_rating, pred_rating):
        for vt, vp in zip(rt, rp):
            if not np.isnan(vt) and vt > 0:
                error += pow(vt - vp, 2)
                count +=1
    return np.sqrt(error / count) if count >0 else 0.0

def mse_validation(self, true_rating, test_rating, pred_rating):
    error = 0.0
    count =0
    for rt, rtt, rp in zip(true_rating, test_rating, pred_rating):
        for vt, vtt, vp in zip(rt, rtt, rp):
            if vt == -1 and not np.isnan(vtt):
                error += pow(vtt - vp, 2)
                count +=1
    return np.sqrt(error / count) if count >0 else 0.0
步骤4:调整超参数范围

将学习率的最大值从0.5改为0.1,避免数值爆炸:

lr = (0.001, 0.01, 0.1)
步骤5:验证数据划分逻辑

确保训练集中标记为-1的是验证样本,且验证集样本无NaN,否则调整mse_validation中的判断条件为np.isnan(vt),直接用测试集的有效评分计算误差。


内容的提问来源于stack exchange,提问作者angryhorse

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.24 21:32:01