You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

构建CNN模型时遇AttributeError与RuntimeError问题求助

问题解决步骤

1. 解决AttributeError: 'Sequential' object has no attribute 'weight'

原代码中self.fnn是nn.Sequential容器,本身没有weight属性,权重属于容器内的nn.Linear层。将初始化代码修改为:

nn.init.xavier_uniform_(self.fnn[0].weight)

这样就能正确访问到全连接层的权重参数。

2. 解决state_dict键不匹配的RuntimeError

这个错误说明你现在加载的模型文件(epoch_10.pt)是用旧版网络结构训练保存的,和当前代码的网络结构参数命名不一致:

  • 当前结构用Sequential包裹卷积和全连接层,参数键名是conv1.0.weight、fnn.0.weight这类格式
  • 旧模型的参数键名是linear1.weight、conv1.weight这类无容器层级的命名

两种解决方式:

方式一:重新训练(推荐)

如果旧模型不是必须保留的,直接用修改后的正确结构重新训练:

  1. 把config里的mode改成"train"
  2. 运行代码,训练完成后会生成新的模型文件,之后测试时加载新生成的模型即可。

方式二:兼容旧模型(手动映射键名)

如果必须加载旧模型,需要手动修改state_dict的键名来匹配当前结构:
在test函数中加载模型时,添加键名映射逻辑:

def test(config):
    model = MNIST_CNN(config).cuda()
    # 加载旧模型的state_dict
    old_state_dict = torch.load(os.path.join(config["output_dir"], config["model_name"]))
    # 构建新的键名映射
    new_state_dict = {}
    for key in old_state_dict.keys():
        if key.startswith("conv1"):
            # 旧键:conv1.weight → 新键:conv1.0.weight
            new_key = key.replace("conv1", "conv1.0")
        elif key.startswith("conv2"):
            new_key = key.replace("conv2", "conv2.0")
        elif key.startswith("linear1"):
            # 旧键:linear1.weight → 新键:fnn.0.weight
            new_key = key.replace("linear1", "fnn.0")
        else:
            new_key = key
        new_state_dict[new_key] = old_state_dict[key]
    # 加载修改后的state_dict
    model.load_state_dict(new_state_dict)
    # 后续代码不变...

注意:如果旧模型的网络结构和当前结构的参数数量不匹配(比如旧模型有两个全连接层,当前只有一个),这种方法会失效,此时只能重新训练。

修改后的完整代码

from google.colab import drive
drive.mount('/gdrive', force_remount=True)

import os
import numpy as np
import torch
import torch.nn as nn
from sklearn.metrics import accuracy_score
from torch.utils.data import (DataLoader, RandomSampler, TensorDataset)
from keras.datasets import mnist

class MNIST_CNN(nn.Module):
  def __init__(self, config):
    super(MNIST_CNN, self).__init__()

    self.conv1 = nn.Sequential(
        nn.Conv2d(1, 32, kernel_size=3, stride=1, padding=1),
        nn.ReLU(),
        nn.MaxPool2d(kernel_size=2, stride=2)
    )

    self.conv2 = nn.Sequential(
        nn.Conv2d(32, 64, kernel_size=3, stride=1, padding=1),
        nn.ReLU(),
        nn.MaxPool2d(kernel_size=2, stride=2)
    )

    self.fnn = nn.Sequential(
        nn.Linear(7*7*64, 10, bias=True)
    )

    # 修正权重初始化
    nn.init.xavier_uniform_(self.fnn[0].weight)

  def forward(self,input_features):
    output = self.conv1(input_features)
    output = self.conv2(output)
    output = output.view(output.size(0), -1)
    hypothesis = self.fnn(output)
    return hypothesis

def load_dataset():
  (train_X, train_Y), (test_X, test_Y) = mnist.load_data()

  train_X = train_X.reshape(-1, 1, 28, 28)
  test_X = test_X.reshape(-1, 1, 28, 28)

  train_X = torch.tensor(train_X, dtype=torch.float)
  train_Y = torch.tensor(train_Y, dtype=torch.long)
  test_X = torch.tensor(test_X, dtype=torch.float)
  test_Y = torch.tensor(test_Y, dtype=torch.long)

  return (train_X, train_Y), (test_X, test_Y)

def tensor2list(input_tensor):
  return input_tensor.cpu().detach().numpy().tolist()

def do_test(model, test_dataloader):
  model.eval()

  predicts, golds = [], []

  with torch.no_grad():
    for step, batch in enumerate(test_dataloader):
      batch = tuple(t.cuda() for t in batch)
      input_features, labels = batch

      hypothesis = model(input_features)
      print("size of hypothesis", hypothesis.size())
      logits = torch.argmax(hypothesis, -1)

      x = tensor2list(logits)
      y = tensor2list(labels)

      predicts.extend(x)
      golds.extend(y)

  print("PRED=", predicts)
  print("GOLD=", golds)
  print("Accuracy={0:f}\n".format(accuracy_score(golds, predicts)))

def test(config):
  model = MNIST_CNN(config).cuda()
  
  # 若需兼容旧模型,替换下面两行为方式二中的键名映射代码
  model_path = os.path.join(config["output_dir"], config["model_name"])
  model.load_state_dict(torch.load(model_path))

  (_, _), (features, labels) = load_dataset()

  test_features = TensorDataset(features, labels)
  test_dataloader = DataLoader(test_features, shuffle=True, batch_size=config["batch_size"])

  do_test(model, test_dataloader)

def train(config):
  model = MNIST_CNN(config).cuda()

  (input_features, labels), (_, _) = load_dataset()

  train_features = TensorDataset(input_features, labels)
  train_dataloader = DataLoader(train_features, shuffle=True, batch_size=config["batch_size"])

  loss_func = nn.CrossEntropyLoss()
  optimizer = torch.optim.Adam(model.parameters(), lr=config["learn_rate"])

  # 修正config中的epochs拼写(原代码是epoch,少了s)
  for epoch in range(config["epochs"]):
    model.train()

    costs = []

    for step, batch in enumerate(train_dataloader):
      batch = tuple(t.cuda() for t in batch)
      input_features, labels = batch

      optimizer.zero_grad()

      hypothesis = model(input_features)
      cost = loss_func(hypothesis, labels)

      cost.backward()
      optimizer.step()

      costs.append(cost.data.item())
    
    print("Average Loss={0:f}".format(np.mean(costs)))
    save_path = os.path.join(config["output_dir"], f"epoch_{epoch+1}.pt")
    torch.save(model.state_dict(), save_path)
    do_test(model, train_dataloader)

if(__name__=="__main__"):
    root_dir = "/gdrive/My Drive/24-2/MachineLearning"
    output_dir = os.path.join(root_dir, "output")
    if not os.path.exists(output_dir):
        os.makedirs(output_dir)

    # 修正config中的epochs键名(原代码是epoch,少了s)
    config = {"mode": "train",
              "model_name":"epoch_{0:d}.pt".format(10),
              "output_dir":output_dir,
              "learn_rate":0.001,
              "batch_size":32,
              "epochs":10,
              }

    if(config["mode"] == "train"):
        train(config)
    else:
        test(config)

另外注意到原代码中config里的键名是"epoch":10,但train函数里用的是config["epochs"],这里也做了修正,避免训练时出现KeyError。


内容的提问来源于stack exchange,提问作者kkk

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.16 13:35:14