You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

TF2.6自定义Bahdanau注意力层报Keras符号张量转numpy数组错误

问题背景

在TensorFlow v2.6、Keras v2.6环境下搭建基于LSTM的Seq2Seq模型,调用自定义Bahdanau注意力层时直接触发报错,暂未找到有效修复方法。

报错信息

运行代码时抛出两类报错:

  • TypeError: 'KerasTensor' object cannot be interpreted as an integer
  • TypeError: Cannot convert a symbolic Keras input/output to a numpy array. This error may indicate that you're trying to pass a symbolic value to a NumPy call, which is not supported. Or, you may be trying to pass Keras symbolic inputs/outputs to a TF API that does not register dispatching, preventing Keras from automatically converting the API call to a lambda layer in the Functional Model.
问题复现代码

自定义Bahdanau注意力层实现

import tensorflow as tf
import os
from tensorflow.python.keras.layers import Layer
from tensorflow.python.keras import backend as K


class AttentionLayer(Layer):
    """
    Bahdanau注意力实现,引入W_a、U_a、V_a三组可训练权重
     """

    def __init__(self, **kwargs):
        super(AttentionLayer, self).__init__(**kwargs)

    def build(self, input_shape):
        assert isinstance(input_shape, list)
        # 为层创建可训练权重变量
        self.W_a = self.add_weight(name='W_a',
                                   shape=tf.TensorShape((input_shape[0][2], input_shape[0][2])),
                                   initializer='uniform',
                                   trainable=True)
        self.U_a = self.add_weight(name='U_a',
                                   shape=tf.TensorShape((input_shape[1][2], input_shape[0][2])),
                                   initializer='uniform',
                                   trainable=True)
        self.V_a = self.add_weight(name='V_a',
                                   shape=tf.TensorShape((input_shape[0][2], 1)),
                                   initializer='uniform',
                                   trainable=True)

        super(AttentionLayer, self).build(input_shape)

    def call(self, inputs, verbose=False):
        """
        inputs: [encoder_output_sequence, decoder_output_sequence]
        """
        assert type(inputs) == list
        encoder_out_seq, decoder_out_seq = inputs
        if verbose:
            print('encoder_out_seq>', encoder_out_seq.shape)
            print('decoder_out_seq>', decoder_out_seq.shape)

        def energy_step(inputs, states):
            """ 计算单个解码器状态对应能量值的步进函数
            inputs: (batchsize * 1 * de_in_dim)
            states: (batchsize * 1 * de_latent_dim)
            """

            assert_msg = "States must be an iterable. Got {} of type {}".format(states, type(states))
            assert isinstance(states, list) or isinstance(states, tuple), assert_msg

            """ 张量形状参数定义"""
            en_seq_len, en_hidden = encoder_out_seq.shape[1], encoder_out_seq.shape[2]
            de_hidden = inputs.shape[-1]

            """ 计算S.Wa,其中S=[s0, s1, ..., si]"""
            W_a_dot_s = K.dot(encoder_out_seq, self.W_a)

            """ 计算hj.Ua """
            U_a_dot_h = K.expand_dims(K.dot(inputs, self.U_a), 1)
            if verbose:
                print('Ua.h>', U_a_dot_h.shape)

            """ 计算tanh(S.Wa + hj.Ua) """
            Ws_plus_Uh = K.tanh(W_a_dot_s + U_a_dot_h)
            if verbose:
                print('Ws+Uh>', Ws_plus_Uh.shape)

            """ 计算softmax(va.tanh(S.Wa + hj.Ua)) """
            e_i = K.squeeze(K.dot(Ws_plus_Uh, self.V_a), axis=-1)
            e_i = K.softmax(e_i)

            if verbose:
                print('ei>', e_i.shape)

            return e_i, [e_i]

        def context_step(inputs, states):
            """ 基于ei计算上下文向量ci的步进函数 """

            assert_msg = "States must be an iterable. Got {} of type {}".format(states, type(states))
            assert isinstance(states, list) or isinstance(states, tuple), assert_msg

            c_i = K.sum(encoder_out_seq * K.expand_dims(inputs, -1), axis=1)
            if verbose:
                print('ci>', c_i.shape)
            return c_i, [c_i]

        fake_state_c = K.sum(encoder_out_seq, axis=1)
        fake_state_e = K.sum(encoder_out_seq, axis=2)

        """ 计算能量输出 """
        last_out, e_outputs, _ = K.rnn(
            energy_step, decoder_out_seq, [fake_state_e],
        )

        """ 计算上下文向量 """
        last_out, c_outputs, _ = K.rnn(
            context_step, e_outputs, [fake_state_c],
        )

        return c_outputs, e_outputs

    def compute_output_shape(self, input_shape):
        """ 层输出形状定义 """
        return [
            tf.TensorShape((input_shape[1][0], input_shape[1][1], input_shape[1][2])),
            tf.TensorShape((input_shape[1][0], input_shape[1][1], input_shape[0][1]))
        ]

编码器-解码器模型构建代码

from keras import backend as K 
K.clear_session() 
latent_dim = 500 

# 编码器结构
encoder_inputs = Input(shape=(max_len_text,)) 
enc_emb = Embedding(x_voc_size, latent_dim,trainable=True)(encoder_inputs) 

#LSTM 层1
encoder_lstm1 = LSTM(latent_dim,return_sequences=True,return_state=True,dropout=0.5) 
encoder_output1, state_h1, state_c1 = encoder_lstm1(enc_emb) 

#LSTM 层2
encoder_lstm2 = LSTM(latent_dim,return_sequences=True,return_state=True,dropout=0.5) 
encoder_output2, state_h2, state_c2 = encoder_lstm2(encoder_output1)

#LSTM 层3
encoder_lstm3=LSTM(latent_dim, return_state=True, return_sequences=True,dropout=0.5) 
encoder_outputs, state_h, state_c= encoder_lstm3(encoder_output2) 

# 解码器结构
decoder_inputs = Input(shape=(None,)) 
dec_emb_layer = Embedding(y_voc_size, latent_dim,trainable=True) 
dec_emb = dec_emb_layer(decoder_inputs) 

# 编码器输出状态作为解码器初始状态
decoder_lstm = LSTM(latent_dim, return_sequences=True, return_state=True) 
decoder_outputs,decoder_fwd_state, decoder_back_state = decoder_lstm(dec_emb,initial_state=[state_h, state_c]) 

# 注意力层调用
attn_layer = AttentionLayer(name='attention_layer') 
attn_out, attn_states = attn_layer([encoder_outputs, decoder_outputs]) 

# 拼接注意力输出和解码器LSTM输出
decoder_concat_input = Concatenate(axis=-1, name='concat_layer')([decoder_outputs, attn_out])

# 全连接输出层
decoder_dense = TimeDistributed(Dense(y_voc_size, activation='softmax')) 
decoder_outputs = decoder_dense(decoder_concat_input) 
报错原因
  1. 代码从tensorflow.python.keras导入Layer和K后端,属于TensorFlow内部私有API,这类API没有适配TF2.x的Keras符号张量(KerasTensor)分发机制,调用时无法自动转换为函数式模型兼容的层,触发符号张量转numpy数组的报错。
  2. 注意力逻辑中直接取encoder_out_seq.shape[1]这类动态维度值,在函数式模型构建阶段拿到的是KerasTensor类型的符号维度,不是Python整数,无法被后续逻辑识别为整数参数,触发第一个类型错误。
  3. 代码使用TF1.x时代的K.rnn步进接口逐时间步计算注意力权重,这类接口在TF2.6下对符号张量的兼容性较差,且计算效率低。
修复方案
  1. 替换所有私有API导入,统一使用tensorflow.keras下的公开接口。
  2. 动态序列长度维度使用tf.shape()获取张量运行时的实际整数值,固定的隐层维度可以直接用静态shape获取。
  3. 移除K.rnn步进逻辑,改用TF2.x支持的广播机制批量计算所有时间步的注意力权重,避免步进接口的兼容性问题,同时提升计算速度。

修复后的注意力层代码如下,替换原有注意力层后无需修改其他模型搭建代码即可正常运行:

import tensorflow as tf
from tensorflow.keras.layers import Layer
import tensorflow.keras.backend as K


class AttentionLayer(Layer):
    """
    适配TF2.6+/Keras2.6+版本的Bahdanau注意力实现
    """
    def __init__(self, **kwargs):
        super(AttentionLayer, self).__init__(**kwargs)

    def build(self, input_shape):
        assert isinstance(input_shape, list)
        enc_shape, dec_shape = input_shape
        enc_hidden = enc_shape[-1]
        dec_hidden = dec_shape[-1]
        
        self.W_a = self.add_weight(name='W_a',
                                   shape=(enc_hidden, enc_hidden),
                                   initializer='uniform',
                                   trainable=True)
        self.U_a = self.add_weight(name='U_a',
                                   shape=(dec_hidden, enc_hidden),
                                   initializer='uniform',
                                   trainable=True)
        self.V_a = self.add_weight(name='V_a',
                                   shape=(enc_hidden, 1),
                                   initializer='uniform',
                                   trainable=True)

        super(AttentionLayer, self).build(input_shape)

    def call(self, inputs, verbose=False):
        assert type(inputs) == list
        encoder_out_seq, decoder_out_seq = inputs
        
        # 计算encoder侧投影:(batch_size, enc_seq_len, enc_hidden)
        W_a_dot_s = K.dot(encoder_out_seq, self.W_a)
        # 计算decoder侧投影:(batch_size, dec_seq_len, enc_hidden)
        U_a_dot_h = K.dot(decoder_out_seq, self.U_a)
        
        # 广播机制计算注意力分数,输出维度(batch_size, dec_seq_len, enc_seq_len, enc_hidden)
        Ws_plus_Uh = K.tanh(tf.expand_dims(W_a_dot_s, 1) + tf.expand_dims(U_a_dot_h, 2))
        
        # 计算注意力权重:(batch_size, dec_seq_len, enc_seq_len)
        e = K.squeeze(K.dot(Ws_plus_Uh, self.V_a), axis=-1)
        e_outputs = K.softmax(e, axis=-1)
        
        # 加权求和得到上下文向量:(batch_size, dec_seq_len, enc_hidden)
        c_outputs = tf.matmul(e_outputs, encoder_out_seq)
        
        if verbose:
            print('encoder_out_seq shape:', encoder_out_seq.shape)
            print('decoder_out_seq shape:', decoder_out_seq.shape)
            print('attention weights shape:', e_outputs.shape)
            print('context vector shape:', c_outputs.shape)
            
        return c_outputs, e_outputs

    def compute_output_shape(self, input_shape):
        enc_shape, dec_shape = input_shape
        return [
            (dec_shape[0], dec_shape[1], enc_shape[-1]),
            (dec_shape[0], dec_shape[1], enc_shape[1])
        ]

内容的提问来源于stack exchange,提问作者Bboss Boss

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.29 16:39:19