You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

PyTorch RuntimeError:二次反向传播报错问题求助

问题:PyTorch RNN字符生成模型反向传播报错

环境信息

  • PyTorch 2.0.0 CPU版本
  • Python 3.10.10 64位

问题现象

运行RNN字符生成代码时触发反向传播错误:RuntimeError: Trying to backward through the graph a second time,尝试添加loss.backward(retain_graph=True)后,又出现原地操作导致的梯度计算错误:RuntimeError: one of the variables needed for gradient computation has been modified by an inplace operation

完整代码

import torch
import torch.nn as nn
import string

# define the vocabulary of characters
vocab = string.ascii_letters + " ."
# define the size of the vocabulary and the hidden state
vocab_size = len(vocab)
hidden_size = 16
# define a mapping from characters to indices and vice versa
char_to_index = {c: i for i, c in enumerate(vocab)}
index_to_char = {i: c for i, c in enumerate(vocab)}

class RNN(nn.Module):

    def __init__(self, n):
        # initialize the parent class
        super(RNN, self).__init__()
        self.n = n
        # define the embedding layer that maps indices to vectors
        self.embedding = nn.Embedding(vocab_size, hidden_size)
        # define the recurrent layer that updates the hidden state
        self.recurrent = nn.Linear(hidden_size, hidden_size)
        # define the output layer that maps hidden state to logits
        self.output = nn.Linear(hidden_size, vocab_size)
        torch.autograd.set_detect_anomaly(True)

    def forward(self, x, h):
        # x is a tensor of shape (self.n) containing indices
        # h is a tensor of shape (1, hidden_size) containing a hidden state
        # embed x into a vector of shape (1, hidden_size)
        x = self.embedding(x)
        # update h with x using a tanh activation function and non-inplace addition
        h_new = torch.tanh(self.recurrent(x).add(h))
        # compute logits from h_new using a linear layer
        logits = self.output(h_new)
        return logits,h_new

    def update(self, text):
        # text is a string containing user input
        # initialize an optimizer and a loss function
        optimizer = torch.optim.SGD(self.parameters(), lr=0.01)
        criterion = nn.CrossEntropyLoss()
        # loop through each character in text except the last self.n ones
        for i in range(len(text) - self.n):
            # get the current and next characters as indices
            current_chars = [char_to_index[c] for c in text[i:i+self.n]]
            next_char = char_to_index[text[i+self.n]]
            # convert them to tensors of shape (self.n) and (1) respectively
            current_chars = torch.tensor(current_chars)
            next_char = torch.tensor([next_char])
            # zero out the gradients from previous step
            optimizer.zero_grad()
            # forward pass through the model and get logits and new hidden state 
            logits, self.h = self.forward(current_chars, self.h)
            # compute loss between logits and next_char 
            loss = criterion(logits.view(1,-1), next_char.view(1))
            print(f"Loss: {loss.item():.4f}")
            # backward pass to compute gradients
            loss.backward()
            # update parameters with gradient descent 
            optimizer.step()
    
    def generate(self, start):
        # start is a string of length self.n to start with
        # get the indices of the start characters
        start_indices = [char_to_index[c] for c in start]
        # convert them to a tensor of shape (self.n)
        start_indices = torch.tensor(start_indices)
        # initialize the output with the start characters
        output = [c for c in start]
        # loop until reaching a period or a maximum length
        while output[-1] != "." and len(output) < 100:
            # forward pass through the model and get logits and new hidden state 
            logits, self.h = self.forward(start_indices, self.h)
            # apply softmax to get probabilities 
            probs = torch.softmax(logits.view(-1), dim=0)
            # sample a next index from the probabilities 
            next_index = torch.multinomial(probs, 1).item()
            # get the next character from the index
            next_char = index_to_char[next_index]
            # append it to the output 
            output.append(next_char)
            # update the start indices with the next index 
            start_indices[:-1] = start_indices[1:]
            start_indices[-1] = next_index
        # join and return the output as a string
        return "".join(output)

if __name__ == '__main__':
    # create a new RNN model with context size
    model = RNN(1)
    # initialize a random hidden state of shape (1, hidden_size)
    model.h = torch.randn(1, hidden_size)
    # update the model with some user input
    model.update("hello world.")
    # generate some text starting with "he"
    print(model.generate("he"))

初始报错信息

Loss: 4.5443
Loss: 4.4064
C:\Users\pythonic\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\torch\autograd\__init__.py:200: UserWarning: Error detected in TanhBackward0. Traceback of forward call that caused the error:
  File "c:\Users\pythonic\Documents\python\bing\markov\v4.py", line 100, in <module>
    model.update("hello world.") #open('english.txt', 'r').read())
  File "c:\Users\pythonic\Documents\python\bing\markov\v4.py", line 57, in update
    logits, self.h = self.forward(current_chars, self.h)
  File "c:\Users\pythonic\Documents\python\bing\markov\v4.py", line 34, in forward
    h_new = torch.tanh(self.recurrent(x).add(h))
 (Triggered internally at ..\torch\csrc\autograd\python_anomaly_mode.cpp:119.)
  Variable._execution_engine.run_backward(  # Calls into the C++ engine to run the backward pass
Traceback (most recent call last):
  File "c:\Users\pythonic\Documents\python\bing\markov\v4.py", line 100, in <module>
    model.update("hello world.") #open('english.txt', 'r').read())
  File "c:\Users\pythonic\Documents\python\bing\markov\v4.py", line 62, in update
    loss.backward()
  File "C:\Users\pythonic\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\torch\_tensor.py", line 487, in backward
    torch.autograd.backward(
  File "C:\Users\pythonic\AppData\Local\Packages\PythonSoftwareFoundation.Python.3.10_qbz5n2kfra8p0\LocalCache\local-packages\Python310\site-packages\torch\autograd\__init__.py", line 200, in backward
    Variable._execution_engine.run_backward(  # Calls into the C++ engine to run the backward pass
RuntimeError: Trying to backward through the graph a second time (or directly access saved tensors after they have already been freed). Saved intermediate values of the graph are freed when you call .backward() or autograd.grad(). Specify retain_graph=True if you need to backward through the graph a second time or if you need to access saved tensors after calling backward.

添加retain_graph=True后的报错

RuntimeError: one of the variables needed for gradient computation has been modified by an inplace operation: [torch.FloatTensor [16, 16]], which is output 0 of TBackward, is at version 2; expected version 1 instead. Hint: the backtrace further above shows the operation that failed to compute its gradient. The variable in question was changed in there or anywhere later. Good luck!

解决方案

错误根源

  1. 隐藏状态绑定计算图:每次迭代中self.h被更新为包含计算图的张量,反向传播后计算图被释放,下一次迭代复用该张量会导致尝试二次反向传播。
  2. 优化器重复初始化:update方法内每次循环都创建新的优化器,导致参数更新无法累积。
  3. 生成阶段原地操作:start_indices[:-1] = start_indices[1:]是原地修改张量,干扰计算图追踪。

修改后的代码

import torch
import torch.nn as nn
import string

# define the vocabulary of characters
vocab = string.ascii_letters + " ."
# define the size of the vocabulary and the hidden state
vocab_size = len(vocab)
hidden_size = 16
# define a mapping from characters to indices and vice versa
char_to_index = {c: i for i, c in enumerate(vocab)}
index_to_char = {i: c for i, c in enumerate(vocab)}

class RNN(nn.Module):

    def __init__(self, n):
        super(RNN, self).__init__()
        self.n = n
        self.embedding = nn.Embedding(vocab_size, hidden_size)
        self.recurrent = nn.Linear(hidden_size, hidden_size)
        self.output = nn.Linear(hidden_size, vocab_size)
        # 初始化隐藏状态,默认不绑定计算图
        self.h = torch.randn(1, hidden_size).detach()

    def forward(self, x, h):
        x = self.embedding(x)
        h_new = torch.tanh(self.recurrent(x) + h)
        logits = self.output(h_new)
        return logits, h_new

    def update(self, text):
        # 优化器只初始化一次
        optimizer = torch.optim.SGD(self.parameters(), lr=0.01)
        criterion = nn.CrossEntropyLoss()
        for i in range(len(text) - self.n):
            current_chars = torch.tensor([char_to_index[c] for c in text[i:i+self.n]])
            next_char = torch.tensor([char_to_index[text[i+self.n]]])
            
            optimizer.zero_grad()
            logits, h_new = self.forward(current_chars, self.h)
            loss = criterion(logits.view(1,-1), next_char)
            print(f"Loss: {loss.item():.4f}")
            # 反向传播
            loss.backward()
            optimizer.step()
            # 分离隐藏状态,断开计算图
            self.h = h_new.detach()
    
    def generate(self, start):
        start_indices = torch.tensor([char_to_index[c] for c in start])
        output = list(start)
        # 生成阶段使用分离的隐藏状态,避免干扰训练计算图
        gen_h = self.h.detach()
        while output[-1] != "." and len(output) < 100:
            logits, gen_h = self.forward(start_indices, gen_h)
            probs = torch.softmax(logits.view(-1), dim=0)
            next_index = torch.multinomial(probs, 1).item()
            next_char = index_to_char[next_index]
            output.append(next_char)
            # 创建新张量替换原张量,避免原地操作
            start_indices = torch.cat([start_indices[1:], torch.tensor([next_index])])
        return "".join(output)

if __name__ == '__main__':
    model = RNN(1)
    model.update("hello world.")
    print(model.generate("he"))

关键修改点

  • 在__init__中初始化self.h并调用.detach(),默认断开计算图。
  • update方法中,每次反向传播后将self.h替换为h_new.detach(),避免后续迭代复用旧计算图。
  • 优化器在update方法开头只初始化一次,确保参数更新累积。
  • generate方法中使用独立的gen_h隐藏状态,不修改模型的self.h,同时用torch.cat创建新张量替换start_indices,避免原地操作。

内容的提问来源于stack exchange,提问作者Pythonic456

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.27 03:57:06