You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C++实现MLP神经网络正常运行出NaN,调试模式正常求助

Eigen3实现MLP时正常运行出现NaN,调试模式正常的问题排查

用Eigen3实现手写数字识别MLP,正常编译运行时权重、偏置、激活值等参数会变为NaN,但VS Code调试模式下完全正常。已尝试关闭编译器优化、检查变量初始化,使用MinGW GCC编译器,求原因分析及解决方法。


网络实现代码

#include "..\Headers\Network.h"
#include <cmath>
#include <iostream>
#include <iterator>
#include <string>
#include <vector>
#include <algorithm>

using std::vector;
using std::string;
using Eigen::VectorXd;
using Eigen::MatrixXd;

double sigmoide (double x);
double sigmoide_derivative(double x);
VectorXd sigmoide_derivative(VectorXd vec);
void load_val (string pi, VectorXd& pixels);
std::ofstream _log ("log.txt");

template <typename T>
void print(T& mat)
{
    for (int r = 0; r < mat.rows(); r++)
    {
        for (int c = 0; c < mat.cols(); c++)
        {
        }
    }
}

Network::Network(vector<string> _data, int _l_rate, vector<int> dim) : data{_data}, l_rate{_l_rate}
{
    layers = dim.size();
    for (int i = 0; i < layers - 1; i++)
    {
        MatrixXd m = MatrixXd::Random(dim[i + 1], dim[i]);
        weights.push_back(m);
        VectorXd b = VectorXd::Random(dim[i + 1]);
        biases.push_back(b);
        VectorXd _z (dim[i + 1]);
        z.push_back(_z);
        VectorXd n (dim[i]);
        neurons.push_back(n);
    }
    VectorXd n_f (dim[layers - 1]);
    neurons.push_back(n_f);
}

void Network::learn(int epoch, int mini_batch)
{

    for(int e = 0; e < epoch; e++)
    {
        std::cout << "Epoch: " << e + 1 << "\n\n";
        double e_cost = 0;
        shuffle(begin(data), end(data), rng);
        for(unsigned long long int n = 0; n < data.size();)
        {
            e_cost += SGD(mini_batch, n);
        }
    }
}


double Network::SGD (int mini_batch, unsigned long long int& n_data)
{
    vector<VectorXd> p_d_biases = vector<VectorXd>();
    vector<MatrixXd> p_d_weights = vector<MatrixXd>();
    double b_cost = 0;
    for(int m = 0; m < mini_batch && n_data != data.size(); m++, n_data++)
    {
        feed_forward(data.at(n_data));
        SGD_b(p_d_biases, data.at(n_data));
        SGD_w(p_d_weights, p_d_biases);
        b_cost += cost(data.at(n_data));
    }
    step(p_d_weights, p_d_biases, mini_batch);
    return b_cost / mini_batch;
}

double Network::cost(string sample)
{
    VectorXd e_values = VectorXd();
    exp_values(sample, e_values);
    double s_cost = 0;
    for(int i = 0; i < neurons[layers - 1].size(); i++)
    {
        if(e_values[i] == 1)
        {
            s_cost += std::pow(neurons.at(layers - 1)(i) - 1, 2);
        }
        else
        {
            s_cost += std::pow(neurons.at(layers - 1)(i), 2);
        }
    }
    return s_cost;
}

void Network::step(const vector<MatrixXd>& p_d_weights, const vector<VectorXd>& p_d_biases, int mini_batch)
{
    for(int l = layers - 2, n = 0; l >= 0; l--, n++)
    {
        VectorXd b_tmp (biases.at(l).rows());
        MatrixXd w_tmp (weights.at(l).rows(), weights.at(l).cols());
        for(int i = 0; i < layers - 1; i++)
        {
            b_tmp += p_d_biases.at((i * (layers - 1)) + n);
            w_tmp += p_d_weights.at((i * (layers - 1)) + n);
        }
        biases.at(l) -= l_rate * (b_tmp / mini_batch);
        weights.at(l) -= l_rate * (w_tmp / mini_batch);
    }
}

void Network::SGD_w (vector<MatrixXd>& p_d_weights, const vector<VectorXd>& p_d_biases)
{
    for(int l = layers - 2; l >= 0; l--)
    {
        p_d_weights.push_back(p_d_biases.at(p_d_biases.size() - (l + 1)) * neurons.at(l).transpose());
    }
}

void Network::SGD_b (vector<VectorXd>& p_d_biases, string sample)
{
    VectorXd e_values = VectorXd();
    exp_values(sample, e_values);
    for(int l = layers - 1; l > 0; l--)
    {
        if(l == (layers - 1))
        {
            p_d_biases.push_back((2*(neurons.at(l) - e_values)).cwiseProduct(sigmoide_derivative(z.at(l-1))));
        }
        else
        {
            VectorXd b =(weights.at(l).transpose() * p_d_biases.at(p_d_biases.size() - 1)).cwiseProduct(sigmoide_derivative(z.at(l - 1)));
            p_d_biases.push_back(b);
        }
    }
}

void Network::feed_forward(string sample)
{
    load_val(sample, neurons[0]);
    for(int l = 0; l < layers - 1; l++)
    {
        z.at(l) = weights.at(l) * neurons.at(l) + biases.at(l);
        for (int i = 0; i < biases.at(l).size(); i++)
        {
            neurons.at(l + 1)(i) = sigmoide(z.at(l)(i));
        }
    }
}

void Network::exp_values (string sample, VectorXd& e_values)
{
    e_values.resize(neurons[layers - 1].size());
    short digit = std::stoi(string(sample.begin(), sample.begin() + 1));
    for(int i = 0;  i < e_values.size(); i++)
    {
        if(i - digit == 0)
        {
            e_values(i) = 1;
        }
        else
        {
            e_values(i) = 0;
        }
    }
}

void load_val (string pi, VectorXd& pixels)
{
    int i = 0;
    for(auto s = pi.begin() + 3, p = pi.begin() + 1; s != pi.end(); s++)
    {
        if(*s == ',')
        {
            double t = std::stoi(std::string(p + 1, s)) / 255.;
            pixels(i) = t;
            p = s;
            i++;
        }
        if(s == (pi.end() - 1))
        {
            double t = std::stoi(std::string(s, pi.end())) / 255.;
            pixels(i) = t;
        }
    }
}

VectorXd sigmoide_derivative(VectorXd vec)
{
    VectorXd result = VectorXd();
    result.resize(vec.size());
    for(int r = 0; r < vec.rows(); r++)
    {
        result(r) = sigmoide_derivative(vec(r));
    }
    return result;
}

double sigmoide_derivative(double x)
{
    return std::exp(x) / std::pow(1 + std::exp(x), 2);
}

double sigmoide (double x)
{
    return 1. / (1 + (1. / std::exp(x)));
}

主函数代码

int main()
{
    ifstream f_data ("..\\csv_files\\mnist_train.csv");
    vector<string> data;
    if(f_data.good())
    {
        while(!(f_data.eof())) 
        {
            string tmp;
            f_data >> tmp;
            data.push_back(tmp);
        }
        vector<int> dim {784, 16, 10};
        Network n (data, 3, dim);
        n.learn(20, 10);
    
        ifstream t_data ("..\\csv_files\\mnist_test.csv");
        string s;
        t_data >> s;
        cout << string(s.begin(), s.begin() + 1) << endl;
        n.feed_forward(s);
        print(n.getNeurons(2));
    }
    return 0;
}

原因分析及解决方法

1. 梯度爆炸与数值溢出

  • 问题点:
    • 用MatrixXd::Random()初始化的权重/偏置范围是[-1,1],对于784输入层的网络,未缩放的初始权重会导致前向传播时z = W*x + b数值过大,std::exp(x)计算溢出为无穷大,进而产生NaN。
    • 学习率设置为3,远高于MLP的合理范围,权重更新幅度过大,快速进入数值不稳定区域。
  • 解决方法:
    • 改用Xavier初始化缩放权重:MatrixXd m = MatrixXd::Random(dim[i+1], dim[i]) / std::sqrt(dim[i]);,避免初始激活值方差过大。
    • 降低学习率至0.1或0.01,逐步调试找到合适值。

2. Sigmoid函数数值稳定性问题

  • 问题点:
    • 当前sigmoide实现为1. / (1 + (1. / std::exp(x))),当x为负且绝对值很大时,1/std::exp(x)趋近于无穷大,分母溢出导致结果异常,反向传播时衍生NaN。
  • 解决方法:
    • 替换为数值稳定的实现:
      double sigmoide(double x) {
          if (x >= 0) {
              return 1.0 / (1.0 + std::exp(-x));
          } else {
              double exp_x = std::exp(x);
              return exp_x / (1.0 + exp_x);
          }
      }
      

3. 未初始化变量的内存异常

  • 问题点:
    • 构造函数中z和neurons的元素是默认构造的VectorXd,未初始化内存,编译优化下可能出现非法访问,调试模式下内存被默认初始化,因此无问题。
  • 解决方法:
    • 构造时直接初始化向量为零:
      VectorXd _z = VectorXd::Zero(dim[i + 1]);
      z.push_back(_z);
      VectorXd n = VectorXd::Zero(dim[i]);
      neurons.push_back(n);
      

4. SGD梯度累加逻辑错误

  • 问题点:
    • step函数中梯度累加的索引计算(i * (layers - 1)) + n会超出p_d_biases和p_d_weights的实际大小,导致内存越界,破坏数据产生NaN。
  • 解决方法:
    • 重构梯度累加逻辑,按样本和层的顺序正确累加:
      void Network::step(const vector<MatrixXd>& p_d_weights, const vector<VectorXd>& p_d_biases, int mini_batch)
      {
          vector<VectorXd> total_biases(layers-1);
          vector<MatrixXd> total_weights(layers-1);
          // 初始化累加器为零
          for(int l = 0; l < layers-1; l++){
              total_biases[l] = VectorXd::Zero(biases[l].size());
              total_weights[l] = MatrixXd::Zero(weights[l].rows(), weights[l].cols());
          }
          int samples_per_batch = p_d_biases.size() / (layers-1);
          
          for(int s = 0; s < samples_per_batch; s++){
              for(int l = 0; l < layers-1; l++){
                  total_biases[l] += p_d_biases[s*(layers-1) + l];
                  total_weights[l] += p_d_weights[s*(layers-1) + l];
              }
          }
          
          for(int l = 0; l < layers-1; l++){
              biases[l] -= l_rate * (total_biases[l] / mini_batch);
              weights[l] -= l_rate * (total_weights[l] / mini_batch);
          }
      }
      

5. 数据读取的边界错误

  • 问题点:
    • 主函数用while(!f_data.eof())读取CSV会导致最后一行重复读取,引入无效数据;load_val函数未做索引越界检查,可能访问pixels的无效位置。
  • 解决方法:
    • 修改数据读取逻辑:
      string tmp;
      while(getline(f_data, tmp)){
          if(!tmp.empty()){
              data.push_back(tmp);
          }
      }
      
    • 在load_val中添加if(i < pixels.size())的索引检查,避免越界访问。

内容的提问来源于stack exchange,提问作者Patroid

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.18 12:55:55