You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

PyTorch网络推理出现尺寸不匹配错误,寻求解决帮助

PyTorch预测时维度不匹配问题排查与解决

问题描述

使用以下代码调用训练好的PyTorch网络进行文本真伪预测时,出现维度不匹配错误:

import pandas as pd
import torch
import torch.nn as nn
import numpy as np
from sklearn.feature_extraction.text import TfidfVectorizer

FC1_WEIGHT = 105282  # features in trained net
MODEL_NAME = '/path/to/net.pth'

# Define the model
class Net(nn.Module):
    def __init__(self, input_dim):
        super(Net, self).__init__()
        self.fc1 = nn.Linear(input_dim, 64)
        self.fc2 = nn.Linear(64, 32)
        self.fc3 = nn.Linear(32, 2)
        self.dropout = nn.Dropout(0.5)
        self.relu = nn.ReLU()

    def forward(self, x):
        x = self.fc1(x)
        x = self.relu(x)
        x = self.dropout(x)
        x = self.fc2(x)
        x = self.relu(x)
        x = self.dropout(x)
        x = self.fc3(x)
        return x

# Load the data
test_df = pd.read_csv('input.csv')

# Replace missing values with an empty string
test_df = test_df.fillna('')

# Load the vectorizer
vectorizer = TfidfVectorizer(stop_words='english')
vectorizer.fit(test_df['title'] + ' ' + test_df['text'])

# Load the model
# model = Net(len(vectorizer.get_feature_names_out()))
model = Net(FC1_WEIGHT)  # features in trained net
model.load_state_dict(torch.load(MODEL_NAME, map_location=torch.device('cpu')))
model.eval()

# Make predictions
with torch.no_grad():
    x_title = test_df['title']
    x_text = test_df['text']
    x = vectorizer.transform(x_title + ' ' + x_text).toarray()
    x = np.pad(x, ((0, 0), (0, 105282 - x.shape[1])), 'constant')
    inputs = torch.FloatTensor(x).resize(2, FC1_WEIGHT)
    outputs = model(inputs)
    _, predicted = torch.max(outputs.data, 1)
    # print(predicted)
    
    # Format output into readable format: 0 for fake, 1 for real
    predicted = str(predicted)
    if '0' in predicted:
        print('Fake')
    else:
        print('Real')

错误信息

Traceback (most recent call last):
  File "/path/to/main.py", line 51, in <module>
    outputs = model(inputs)
  File "/Library/Frameworks/Python.framework/Versions/3.10/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1194, in _call_impl
    return forward_call(*input, **kwargs)
  File "/path/to/main.py", line 20, in forward
    x = self.fc1(x)
  File "/Library/Frameworks/Python.framework/Versions/3.10/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1194, in _call_impl
    return forward_call(*input, **kwargs)
  File "/Library/Frameworks/Python.framework/Versions/3.10/lib/python3.10/site-packages/torch/nn/modules/linear.py", line 114, in forward
    return F.linear(input, self.weight, self.bias)
RuntimeError: mat1 and mat2 shapes cannot be multiplied (2x633 and 105282x64)

问题原因

  • TF-IDF向量器不匹配:测试时重新对测试数据调用fit(),生成的特征维度(633)和训练时的特征维度(105282)完全不同,导致输入特征维度与模型第一层线性层的输入维度不匹配。
  • 错误的张量形状修改:使用resize(2, FC1_WEIGHT)强制修改张量形状,破坏了数据原有结构,不是正确的维度对齐方式。
  • 预测结果判断逻辑错误:将张量转成字符串判断是否包含'0'的方式不可靠,无法正确处理批量预测场景。

解决步骤

1. 训练时保存TF-IDF向量器

训练阶段必须保存训练好的向量器,确保测试时使用相同的特征空间:

import joblib
# 训练完成后保存向量器
joblib.dump(vectorizer, 'tfidf_vectorizer.pkl')

2. 测试时加载保存的向量器

替换测试代码中重新fit向量器的逻辑,改为加载训练时保存的向量器:

# 替换原有的向量器初始化和fit代码
import joblib
vectorizer = joblib.load('tfidf_vectorizer.pkl')

3. 修正张量处理和预测结果判断

去掉错误的resize操作,直接使用向量器生成的特征转成张量;同时修正预测结果的判断逻辑:

# 修正后的预测代码
with torch.no_grad():
    combined_text = test_df['title'] + ' ' + test_df['text']
    x = vectorizer.transform(combined_text).toarray()
    inputs = torch.FloatTensor(x)
    outputs = model(inputs)
    _, predicted = torch.max(outputs.data, 1)
    
    # 批量输出每个样本的结果
    for idx, pred in enumerate(predicted):
        print(f"样本{idx+1}: {'Fake' if pred.item() == 0 else 'Real'}")

完整修正代码

import pandas as pd
import torch
import torch.nn as nn
import numpy as np
import joblib
from sklearn.feature_extraction.text import TfidfVectorizer

FC1_WEIGHT = 105282  # features in trained net
MODEL_NAME = '/path/to/net.pth'
VECTORIZER_PATH = 'tfidf_vectorizer.pkl'

# Define the model
class Net(nn.Module):
    def __init__(self, input_dim):
        super(Net, self).__init__()
        self.fc1 = nn.Linear(input_dim, 64)
        self.fc2 = nn.Linear(64, 32)
        self.fc3 = nn.Linear(32, 2)
        self.dropout = nn.Dropout(0.5)
        self.relu = nn.ReLU()

    def forward(self, x):
        x = self.fc1(x)
        x = self.relu(x)
        x = self.dropout(x)
        x = self.fc2(x)
        x = self.relu(x)
        x = self.dropout(x)
        x = self.fc3(x)
        return x

# Load the data
test_df = pd.read_csv('input.csv')
test_df = test_df.fillna('')

# Load the pre-trained vectorizer
vectorizer = joblib.load(VECTORIZER_PATH)

# Load the model
model = Net(FC1_WEIGHT)
model.load_state_dict(torch.load(MODEL_NAME, map_location=torch.device('cpu')))
model.eval()

# Make predictions
with torch.no_grad():
    combined_text = test_df['title'] + ' ' + test_df['text']
    x = vectorizer.transform(combined_text).toarray()
    inputs = torch.FloatTensor(x)
    outputs = model(inputs)
    _, predicted = torch.max(outputs.data, 1)
    
    # 输出每个样本的结果
    for idx, pred in enumerate(predicted):
        print(f"样本{idx+1}: {'Fake' if pred.item() == 0 else 'Real'}")

内容的提问来源于stack exchange,提问作者Boo Who

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.28 23:27:21