在模型中集成分词时遇ValueError: as_list()未定义于未知TensorShape
解决TensorFlow自定义分词层引发的
as_list() is not defined on an unknown TensorShape错误 问题背景
我尝试将分词操作集成到TensorFlow模型中,以此减少CPU占用与内存消耗、提升GPU利用率,但运行时触发错误:
ValueError: as_list() is not defined on an unknown TensorShape.
自定义分词层代码如下:
class TokenizationLayer(Layer): def __init__(self, max_length, **kwargs): super(TokenizationLayer, self).__init__(**kwargs) self.max_length = max_length self.tokenizer = Tokenizer() def build(self, input_shape): super(TokenizationLayer, self).build(input_shape) def tokenize_sequences(self, x): # Tokenization function return self.tokenizer.texts_to_sequences([x.numpy()])[0] def call(self, inputs): # Use tf.py_function to apply tokenization element-wise sequences = tf.map_fn(lambda x: tf.py_function(self.tokenize_sequences, [x], tf.int32), inputs, dtype=tf.int32) # Masking step mask = tf.math.logical_not(tf.math.equal(sequences, 0)) return tf.where(mask, sequences, -1) # Using -1 as a mask value def compute_output_shape(self, input_shape): return (input_shape[0], self.max_length) # Use self.max_length instead of trying to access shape
完整模型代码及报错栈如下:
import tensorflow as tf from tensorflow.keras.layers import Layer, Input, Embedding, LSTM, Dense, Concatenate from tensorflow.keras.models import Model from tensorflow.keras.preprocessing.text import Tokenizer from tensorflow.keras.preprocessing.sequence import pad_sequences class TokenizationLayer(Layer): def __init__(self, max_length, **kwargs): super(TokenizationLayer, self).__init__(**kwargs) self.max_length = max_length self.tokenizer = Tokenizer() def build(self, input_shape): super(TokenizationLayer, self).build(input_shape) def tokenize_sequences(self, x): # Tokenization function return self.tokenizer.texts_to_sequences([x.numpy()])[0] def call(self, inputs): # Use tf.py_function to apply tokenization element-wise sequences = tf.map_fn(lambda x: tf.py_function(self.tokenize_sequences, [x], tf.int32), inputs, dtype=tf.int32) # Masking step mask = tf.math.logical_not(tf.math.equal(sequences, 0)) return tf.where(mask, sequences, -1) # Using -1 as a mask value def compute_output_shape(self, input_shape): return (input_shape[0], self.max_length) # Use self.max_length instead of trying to access shape # Build the model with the custom tokenization layer def build_model(vocab_size, max_length): input1 = Input(shape=(1,), dtype=tf.string) input2 = Input(shape=(1,), dtype=tf.string) # Tokenization layer tokenization_layer = TokenizationLayer(max_length) embedded_seq1 = tokenization_layer(input1) embedded_seq2 = tokenization_layer(input2) # Embedding layer for encoding strings embedding_layer = Embedding(input_dim=vocab_size, output_dim=128, input_length=max_length) # Encode first string lstm_out1 = LSTM(64)(embedding_layer(embedded_seq1)) # Encode second string lstm_out2 = LSTM(64)(embedding_layer(embedded_seq2)) # Concatenate outputs concatenated = Concatenate()([lstm_out1, lstm_out2]) # Dense layer for final output output = Dense(1, activation='relu')(concatenated) # Build model model = Model(inputs=[input1, input2], outputs=output) return model string1 = "hello world" string2 = "foo bar baz" max_length = max(len(string1.split()), len(string2.split())) model = build_model(vocab_size=1000, max_length=max_length) model.summary() labels = tf.random.normal((1, 5)) model.compile(optimizer='adam', loss='mse') model.fit([tf.constant([string1]), tf.constant([string2])], labels, epochs=10, batch_size=1, validation_split=0.2)
报错栈:
WARNING:tensorflow:From /usr/local/lib/python3.10/dist-packages/tensorflow/python/util/deprecation.py:660: calling map_fn_v2 (from tensorflow.python.ops.map_fn) with dtype is deprecated and will be removed in a future version. Instructions for updating: Use fn_output_signature instead --------------------------------------------------------------------------- ValueError Traceback (most recent call last) <ipython-input-1-23051fe36790> in <cell line: 64>() 62 max_length = max(len(string1.split()), len(string2.split())) 63 ---> 64 model = build_model(vocab_size=1000, max_length=max_length) 65 model.summary() 66 2 frames <ipython-input-1-23051fe36790> in build_model(vocab_size, max_length) 42 43 # Encode first string ---> 44 lstm_out1 = LSTM(64)(embedding_layer(embedded_seq1)) 45 46 # Encode second string /usr/local/lib/python3.10/dist-packages/keras/src/utils/traceback_utils.py in error_handler(*args, **kwargs) 68 # To get the full stack trace, call: 69 # `tf.debugging.disable_traceback_filtering()` ---> 70 raise e.with_traceback(filtered_tb) from None 71 finally: 72 del filtered_tb /usr/local/lib/python3.10/dist-packages/tensorflow/python/framework/tensor_shape.py in as_list(self) 1438 """ 1439 if self._dims is None: -> 1440 raise ValueError("as_list() is not defined on an unknown TensorShape.") 1441 return list(self._dims) 1442 ValueError: as_list() is not defined on an unknown TensorShape.
错误原因
- 静态形状无法推断:
tf.py_function脱离TensorFlow计算图追踪,导致tf.map_fn返回的张量形状未知,而Embedding、LSTM层依赖明确的静态形状初始化参数 - Tokenizer未拟合:代码中创建的Tokenizer从未用文本拟合,调用
texts_to_sequences会返回空序列 - 序列长度不固定:分词后未做pad操作,输出序列长度不一致
- 参数过时:
tf.map_fn使用了已废弃的dtype参数,官方要求改用fn_output_signature
解决方案
核心修改点
- 提前拟合Tokenizer,确保分词有效
- 在自定义层中固定输出形状,让TensorFlow能静态推断
- 对分词结果做pad操作,保证输出维度一致
- 替换过时的
tf.map_fn参数
修正后的完整代码
import tensorflow as tf from tensorflow.keras.layers import Layer, Input, Embedding, LSTM, Dense, Concatenate from tensorflow.keras.models import Model from tensorflow.keras.preprocessing.text import Tokenizer from tensorflow.keras.preprocessing.sequence import pad_sequences class TokenizationLayer(Layer): def __init__(self, tokenizer, max_length, **kwargs): super(TokenizationLayer, self).__init__(**kwargs) self.max_length = max_length self.tokenizer = tokenizer # 传入已拟合的Tokenizer def build(self, input_shape): super(TokenizationLayer, self).build(input_shape) def tokenize_sequences(self, x): # 解码字符串并分词,随后pad到固定长度 text = x.numpy().decode('utf-8') seq = self.tokenizer.texts_to_sequences([text])[0] padded_seq = pad_sequences([seq], maxlen=self.max_length, padding='post', truncating='post')[0] return padded_seq.astype('int32') def call(self, inputs): # 使用fn_output_signature指定输出形状,替代过时的dtype参数 sequences = tf.map_fn( lambda x: tf.py_function( self.tokenize_sequences, [x], tf.int32 ), inputs, fn_output_signature=tf.TensorSpec(shape=(self.max_length,), dtype=tf.int32) ) # 用0作为padding标记,后续Embedding层可自动处理mask return sequences def compute_output_shape(self, input_shape): return (input_shape[0], self.max_length) def build_model(vocab_size, max_length, tokenizer): input1 = Input(shape=(1,), dtype=tf.string) input2 = Input(shape=(1,), dtype=tf.string) tokenization_layer = TokenizationLayer(tokenizer, max_length) embedded_seq1 = tokenization_layer(input1) embedded_seq2 = tokenization_layer(input2) # 开启mask_zero自动处理padding的0值 embedding_layer = Embedding( input_dim=vocab_size, output_dim=128, input_length=max_length, mask_zero=True ) lstm_out1 = LSTM(64)(embedding_layer(embedded_seq1)) lstm_out2 = LSTM(64)(embedding_layer(embedded_seq2)) concatenated = Concatenate()([lstm_out1, lstm_out2]) output = Dense(1, activation='relu')(concatenated) model = Model(inputs=[input1, input2], outputs=output) return model # 提前用训练文本拟合Tokenizer string1 = "hello world" string2 = "foo bar baz" train_texts = [string1, string2] tokenizer = Tokenizer(num_words=1000) tokenizer.fit_on_texts(train_texts) max_length = max(len(string1.split()), len(string2.split())) model = build_model(vocab_size=1000, max_length=max_length, tokenizer=tokenizer) model.summary() # 修正标签形状,与模型输出维度匹配 labels = tf.random.normal((1,)) model.compile(optimizer='adam', loss='mse') model.fit([tf.constant([string1]), tf.constant([string2])], labels, epochs=10, batch_size=1)
内容的提问来源于stack exchange,提问作者Maifee Ul Asad
相关产品推荐
相关产品推荐

