You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

在模型中集成分词时遇ValueError: as_list()未定义于未知TensorShape

解决TensorFlow自定义分词层引发的as_list() is not defined on an unknown TensorShape错误

问题背景

我尝试将分词操作集成到TensorFlow模型中,以此减少CPU占用与内存消耗、提升GPU利用率,但运行时触发错误:

ValueError: as_list() is not defined on an unknown TensorShape.

自定义分词层代码如下:

class TokenizationLayer(Layer):
    def __init__(self, max_length, **kwargs):
        super(TokenizationLayer, self).__init__(**kwargs)
        self.max_length = max_length
        self.tokenizer = Tokenizer()

    def build(self, input_shape):
        super(TokenizationLayer, self).build(input_shape)

    def tokenize_sequences(self, x):
        # Tokenization function
        return self.tokenizer.texts_to_sequences([x.numpy()])[0]

    def call(self, inputs):
        # Use tf.py_function to apply tokenization element-wise
        sequences = tf.map_fn(lambda x: tf.py_function(self.tokenize_sequences, [x], tf.int32), inputs, dtype=tf.int32)
        # Masking step
        mask = tf.math.logical_not(tf.math.equal(sequences, 0))
        return tf.where(mask, sequences, -1)  # Using -1 as a mask value

    def compute_output_shape(self, input_shape):
        return (input_shape[0], self.max_length)  # Use self.max_length instead of trying to access shape

完整模型代码及报错栈如下:

import tensorflow as tf
from tensorflow.keras.layers import Layer, Input, Embedding, LSTM, Dense, Concatenate
from tensorflow.keras.models import Model
from tensorflow.keras.preprocessing.text import Tokenizer
from tensorflow.keras.preprocessing.sequence import pad_sequences

class TokenizationLayer(Layer):
    def __init__(self, max_length, **kwargs):
        super(TokenizationLayer, self).__init__(**kwargs)
        self.max_length = max_length
        self.tokenizer = Tokenizer()

    def build(self, input_shape):
        super(TokenizationLayer, self).build(input_shape)

    def tokenize_sequences(self, x):
        # Tokenization function
        return self.tokenizer.texts_to_sequences([x.numpy()])[0]

    def call(self, inputs):
        # Use tf.py_function to apply tokenization element-wise
        sequences = tf.map_fn(lambda x: tf.py_function(self.tokenize_sequences, [x], tf.int32), inputs, dtype=tf.int32)
        # Masking step
        mask = tf.math.logical_not(tf.math.equal(sequences, 0))
        return tf.where(mask, sequences, -1)  # Using -1 as a mask value

    def compute_output_shape(self, input_shape):
        return (input_shape[0], self.max_length)  # Use self.max_length instead of trying to access shape

# Build the model with the custom tokenization layer
def build_model(vocab_size, max_length):
    input1 = Input(shape=(1,), dtype=tf.string)
    input2 = Input(shape=(1,), dtype=tf.string)

    # Tokenization layer
    tokenization_layer = TokenizationLayer(max_length)
    embedded_seq1 = tokenization_layer(input1)
    embedded_seq2 = tokenization_layer(input2)

    # Embedding layer for encoding strings
    embedding_layer = Embedding(input_dim=vocab_size, output_dim=128, input_length=max_length)

    # Encode first string
    lstm_out1 = LSTM(64)(embedding_layer(embedded_seq1))

    # Encode second string
    lstm_out2 = LSTM(64)(embedding_layer(embedded_seq2))

    # Concatenate outputs
    concatenated = Concatenate()([lstm_out1, lstm_out2])

    # Dense layer for final output
    output = Dense(1, activation='relu')(concatenated)

    # Build model
    model = Model(inputs=[input1, input2], outputs=output)
    return model

string1 = "hello world"
string2 = "foo bar baz"

max_length = max(len(string1.split()), len(string2.split()))

model = build_model(vocab_size=1000, max_length=max_length)
model.summary()

labels = tf.random.normal((1, 5))
model.compile(optimizer='adam', loss='mse')
model.fit([tf.constant([string1]), tf.constant([string2])], labels, epochs=10, batch_size=1, validation_split=0.2)

报错栈:

WARNING:tensorflow:From /usr/local/lib/python3.10/dist-packages/tensorflow/python/util/deprecation.py:660: calling map_fn_v2 (from tensorflow.python.ops.map_fn) with dtype is deprecated and will be removed in a future version.
Instructions for updating:
Use fn_output_signature instead
---------------------------------------------------------------------------
ValueError                                Traceback (most recent call last)
<ipython-input-1-23051fe36790> in <cell line: 64>()
     62 max_length = max(len(string1.split()), len(string2.split()))
     63 
---> 64 model = build_model(vocab_size=1000, max_length=max_length)
     65 model.summary()
     66 

2 frames
<ipython-input-1-23051fe36790> in build_model(vocab_size, max_length)
     42 
     43     # Encode first string
---> 44     lstm_out1 = LSTM(64)(embedding_layer(embedded_seq1))
     45 
     46     # Encode second string

/usr/local/lib/python3.10/dist-packages/keras/src/utils/traceback_utils.py in error_handler(*args, **kwargs)
     68             # To get the full stack trace, call:
     69             # `tf.debugging.disable_traceback_filtering()`
---> 70             raise e.with_traceback(filtered_tb) from None
     71         finally:
     72             del filtered_tb

/usr/local/lib/python3.10/dist-packages/tensorflow/python/framework/tensor_shape.py in as_list(self)
   1438     """
   1439     if self._dims is None:
-> 1440       raise ValueError("as_list() is not defined on an unknown TensorShape.")
   1441     return list(self._dims)
   1442 

ValueError: as_list() is not defined on an unknown TensorShape.

错误原因

  1. 静态形状无法推断:tf.py_function脱离TensorFlow计算图追踪,导致tf.map_fn返回的张量形状未知,而Embedding、LSTM层依赖明确的静态形状初始化参数
  2. Tokenizer未拟合:代码中创建的Tokenizer从未用文本拟合,调用texts_to_sequences会返回空序列
  3. 序列长度不固定:分词后未做pad操作,输出序列长度不一致
  4. 参数过时:tf.map_fn使用了已废弃的dtype参数,官方要求改用fn_output_signature

解决方案

核心修改点

  • 提前拟合Tokenizer,确保分词有效
  • 在自定义层中固定输出形状,让TensorFlow能静态推断
  • 对分词结果做pad操作,保证输出维度一致
  • 替换过时的tf.map_fn参数

修正后的完整代码

import tensorflow as tf
from tensorflow.keras.layers import Layer, Input, Embedding, LSTM, Dense, Concatenate
from tensorflow.keras.models import Model
from tensorflow.keras.preprocessing.text import Tokenizer
from tensorflow.keras.preprocessing.sequence import pad_sequences

class TokenizationLayer(Layer):
    def __init__(self, tokenizer, max_length, **kwargs):
        super(TokenizationLayer, self).__init__(**kwargs)
        self.max_length = max_length
        self.tokenizer = tokenizer  # 传入已拟合的Tokenizer

    def build(self, input_shape):
        super(TokenizationLayer, self).build(input_shape)

    def tokenize_sequences(self, x):
        # 解码字符串并分词,随后pad到固定长度
        text = x.numpy().decode('utf-8')
        seq = self.tokenizer.texts_to_sequences([text])[0]
        padded_seq = pad_sequences([seq], maxlen=self.max_length, padding='post', truncating='post')[0]
        return padded_seq.astype('int32')

    def call(self, inputs):
        # 使用fn_output_signature指定输出形状,替代过时的dtype参数
        sequences = tf.map_fn(
            lambda x: tf.py_function(
                self.tokenize_sequences, 
                [x], 
                tf.int32
            ),
            inputs,
            fn_output_signature=tf.TensorSpec(shape=(self.max_length,), dtype=tf.int32)
        )
        # 用0作为padding标记,后续Embedding层可自动处理mask
        return sequences

    def compute_output_shape(self, input_shape):
        return (input_shape[0], self.max_length)

def build_model(vocab_size, max_length, tokenizer):
    input1 = Input(shape=(1,), dtype=tf.string)
    input2 = Input(shape=(1,), dtype=tf.string)

    tokenization_layer = TokenizationLayer(tokenizer, max_length)
    embedded_seq1 = tokenization_layer(input1)
    embedded_seq2 = tokenization_layer(input2)

    # 开启mask_zero自动处理padding的0值
    embedding_layer = Embedding(
        input_dim=vocab_size, 
        output_dim=128, 
        input_length=max_length,
        mask_zero=True
    )

    lstm_out1 = LSTM(64)(embedding_layer(embedded_seq1))
    lstm_out2 = LSTM(64)(embedding_layer(embedded_seq2))

    concatenated = Concatenate()([lstm_out1, lstm_out2])
    output = Dense(1, activation='relu')(concatenated)

    model = Model(inputs=[input1, input2], outputs=output)
    return model

# 提前用训练文本拟合Tokenizer
string1 = "hello world"
string2 = "foo bar baz"
train_texts = [string1, string2]

tokenizer = Tokenizer(num_words=1000)
tokenizer.fit_on_texts(train_texts)

max_length = max(len(string1.split()), len(string2.split()))

model = build_model(vocab_size=1000, max_length=max_length, tokenizer=tokenizer)
model.summary()

# 修正标签形状,与模型输出维度匹配
labels = tf.random.normal((1,))
model.compile(optimizer='adam', loss='mse')
model.fit([tf.constant([string1]), tf.constant([string2])], labels, epochs=10, batch_size=1)

内容的提问来源于stack exchange,提问作者Maifee Ul Asad

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.28 04:52:32