TF2.6自定义Bahdanau注意力层报Keras符号张量转numpy数组错误
问题背景
在TensorFlow v2.6、Keras v2.6环境下搭建基于LSTM的Seq2Seq模型,调用自定义Bahdanau注意力层时直接触发报错,暂未找到有效修复方法。
报错信息
运行代码时抛出两类报错:
TypeError: 'KerasTensor' object cannot be interpreted as an integerTypeError: Cannot convert a symbolic Keras input/output to a numpy array. This error may indicate that you're trying to pass a symbolic value to a NumPy call, which is not supported. Or, you may be trying to pass Keras symbolic inputs/outputs to a TF API that does not register dispatching, preventing Keras from automatically converting the API call to a lambda layer in the Functional Model.
问题复现代码
自定义Bahdanau注意力层实现
import tensorflow as tf import os from tensorflow.python.keras.layers import Layer from tensorflow.python.keras import backend as K class AttentionLayer(Layer): """ Bahdanau注意力实现,引入W_a、U_a、V_a三组可训练权重 """ def __init__(self, **kwargs): super(AttentionLayer, self).__init__(**kwargs) def build(self, input_shape): assert isinstance(input_shape, list) # 为层创建可训练权重变量 self.W_a = self.add_weight(name='W_a', shape=tf.TensorShape((input_shape[0][2], input_shape[0][2])), initializer='uniform', trainable=True) self.U_a = self.add_weight(name='U_a', shape=tf.TensorShape((input_shape[1][2], input_shape[0][2])), initializer='uniform', trainable=True) self.V_a = self.add_weight(name='V_a', shape=tf.TensorShape((input_shape[0][2], 1)), initializer='uniform', trainable=True) super(AttentionLayer, self).build(input_shape) def call(self, inputs, verbose=False): """ inputs: [encoder_output_sequence, decoder_output_sequence] """ assert type(inputs) == list encoder_out_seq, decoder_out_seq = inputs if verbose: print('encoder_out_seq>', encoder_out_seq.shape) print('decoder_out_seq>', decoder_out_seq.shape) def energy_step(inputs, states): """ 计算单个解码器状态对应能量值的步进函数 inputs: (batchsize * 1 * de_in_dim) states: (batchsize * 1 * de_latent_dim) """ assert_msg = "States must be an iterable. Got {} of type {}".format(states, type(states)) assert isinstance(states, list) or isinstance(states, tuple), assert_msg """ 张量形状参数定义""" en_seq_len, en_hidden = encoder_out_seq.shape[1], encoder_out_seq.shape[2] de_hidden = inputs.shape[-1] """ 计算S.Wa,其中S=[s0, s1, ..., si]""" W_a_dot_s = K.dot(encoder_out_seq, self.W_a) """ 计算hj.Ua """ U_a_dot_h = K.expand_dims(K.dot(inputs, self.U_a), 1) if verbose: print('Ua.h>', U_a_dot_h.shape) """ 计算tanh(S.Wa + hj.Ua) """ Ws_plus_Uh = K.tanh(W_a_dot_s + U_a_dot_h) if verbose: print('Ws+Uh>', Ws_plus_Uh.shape) """ 计算softmax(va.tanh(S.Wa + hj.Ua)) """ e_i = K.squeeze(K.dot(Ws_plus_Uh, self.V_a), axis=-1) e_i = K.softmax(e_i) if verbose: print('ei>', e_i.shape) return e_i, [e_i] def context_step(inputs, states): """ 基于ei计算上下文向量ci的步进函数 """ assert_msg = "States must be an iterable. Got {} of type {}".format(states, type(states)) assert isinstance(states, list) or isinstance(states, tuple), assert_msg c_i = K.sum(encoder_out_seq * K.expand_dims(inputs, -1), axis=1) if verbose: print('ci>', c_i.shape) return c_i, [c_i] fake_state_c = K.sum(encoder_out_seq, axis=1) fake_state_e = K.sum(encoder_out_seq, axis=2) """ 计算能量输出 """ last_out, e_outputs, _ = K.rnn( energy_step, decoder_out_seq, [fake_state_e], ) """ 计算上下文向量 """ last_out, c_outputs, _ = K.rnn( context_step, e_outputs, [fake_state_c], ) return c_outputs, e_outputs def compute_output_shape(self, input_shape): """ 层输出形状定义 """ return [ tf.TensorShape((input_shape[1][0], input_shape[1][1], input_shape[1][2])), tf.TensorShape((input_shape[1][0], input_shape[1][1], input_shape[0][1])) ]
编码器-解码器模型构建代码
from keras import backend as K K.clear_session() latent_dim = 500 # 编码器结构 encoder_inputs = Input(shape=(max_len_text,)) enc_emb = Embedding(x_voc_size, latent_dim,trainable=True)(encoder_inputs) #LSTM 层1 encoder_lstm1 = LSTM(latent_dim,return_sequences=True,return_state=True,dropout=0.5) encoder_output1, state_h1, state_c1 = encoder_lstm1(enc_emb) #LSTM 层2 encoder_lstm2 = LSTM(latent_dim,return_sequences=True,return_state=True,dropout=0.5) encoder_output2, state_h2, state_c2 = encoder_lstm2(encoder_output1) #LSTM 层3 encoder_lstm3=LSTM(latent_dim, return_state=True, return_sequences=True,dropout=0.5) encoder_outputs, state_h, state_c= encoder_lstm3(encoder_output2) # 解码器结构 decoder_inputs = Input(shape=(None,)) dec_emb_layer = Embedding(y_voc_size, latent_dim,trainable=True) dec_emb = dec_emb_layer(decoder_inputs) # 编码器输出状态作为解码器初始状态 decoder_lstm = LSTM(latent_dim, return_sequences=True, return_state=True) decoder_outputs,decoder_fwd_state, decoder_back_state = decoder_lstm(dec_emb,initial_state=[state_h, state_c]) # 注意力层调用 attn_layer = AttentionLayer(name='attention_layer') attn_out, attn_states = attn_layer([encoder_outputs, decoder_outputs]) # 拼接注意力输出和解码器LSTM输出 decoder_concat_input = Concatenate(axis=-1, name='concat_layer')([decoder_outputs, attn_out]) # 全连接输出层 decoder_dense = TimeDistributed(Dense(y_voc_size, activation='softmax')) decoder_outputs = decoder_dense(decoder_concat_input)
报错原因
- 代码从
tensorflow.python.keras导入Layer和K后端,属于TensorFlow内部私有API,这类API没有适配TF2.x的Keras符号张量(KerasTensor)分发机制,调用时无法自动转换为函数式模型兼容的层,触发符号张量转numpy数组的报错。 - 注意力逻辑中直接取
encoder_out_seq.shape[1]这类动态维度值,在函数式模型构建阶段拿到的是KerasTensor类型的符号维度,不是Python整数,无法被后续逻辑识别为整数参数,触发第一个类型错误。 - 代码使用TF1.x时代的
K.rnn步进接口逐时间步计算注意力权重,这类接口在TF2.6下对符号张量的兼容性较差,且计算效率低。
修复方案
- 替换所有私有API导入,统一使用
tensorflow.keras下的公开接口。 - 动态序列长度维度使用
tf.shape()获取张量运行时的实际整数值,固定的隐层维度可以直接用静态shape获取。 - 移除
K.rnn步进逻辑,改用TF2.x支持的广播机制批量计算所有时间步的注意力权重,避免步进接口的兼容性问题,同时提升计算速度。
修复后的注意力层代码如下,替换原有注意力层后无需修改其他模型搭建代码即可正常运行:
import tensorflow as tf from tensorflow.keras.layers import Layer import tensorflow.keras.backend as K class AttentionLayer(Layer): """ 适配TF2.6+/Keras2.6+版本的Bahdanau注意力实现 """ def __init__(self, **kwargs): super(AttentionLayer, self).__init__(**kwargs) def build(self, input_shape): assert isinstance(input_shape, list) enc_shape, dec_shape = input_shape enc_hidden = enc_shape[-1] dec_hidden = dec_shape[-1] self.W_a = self.add_weight(name='W_a', shape=(enc_hidden, enc_hidden), initializer='uniform', trainable=True) self.U_a = self.add_weight(name='U_a', shape=(dec_hidden, enc_hidden), initializer='uniform', trainable=True) self.V_a = self.add_weight(name='V_a', shape=(enc_hidden, 1), initializer='uniform', trainable=True) super(AttentionLayer, self).build(input_shape) def call(self, inputs, verbose=False): assert type(inputs) == list encoder_out_seq, decoder_out_seq = inputs # 计算encoder侧投影:(batch_size, enc_seq_len, enc_hidden) W_a_dot_s = K.dot(encoder_out_seq, self.W_a) # 计算decoder侧投影:(batch_size, dec_seq_len, enc_hidden) U_a_dot_h = K.dot(decoder_out_seq, self.U_a) # 广播机制计算注意力分数,输出维度(batch_size, dec_seq_len, enc_seq_len, enc_hidden) Ws_plus_Uh = K.tanh(tf.expand_dims(W_a_dot_s, 1) + tf.expand_dims(U_a_dot_h, 2)) # 计算注意力权重:(batch_size, dec_seq_len, enc_seq_len) e = K.squeeze(K.dot(Ws_plus_Uh, self.V_a), axis=-1) e_outputs = K.softmax(e, axis=-1) # 加权求和得到上下文向量:(batch_size, dec_seq_len, enc_hidden) c_outputs = tf.matmul(e_outputs, encoder_out_seq) if verbose: print('encoder_out_seq shape:', encoder_out_seq.shape) print('decoder_out_seq shape:', decoder_out_seq.shape) print('attention weights shape:', e_outputs.shape) print('context vector shape:', c_outputs.shape) return c_outputs, e_outputs def compute_output_shape(self, input_shape): enc_shape, dec_shape = input_shape return [ (dec_shape[0], dec_shape[1], enc_shape[-1]), (dec_shape[0], dec_shape[1], enc_shape[1]) ]
内容的提问来源于stack exchange,提问作者Bboss Boss
相关产品推荐
相关产品推荐

