You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

TensorFlow2.16+Keras3.0下BiLSTM-CRF命名实体识别替代方案求助

问题

在TensorFlow 2.16 + Keras 3.0环境下构建BiLSTM-CRF模型完成命名实体识别(NER)任务时,使用已废弃的keras_contrib或tensorflow_addons提供的CRF层出现严重兼容性问题(如运行时报错module 'keras.backend' has no attribute 'dot'),且不愿降级TensorFlow版本,求可行的替代方案。

原实现代码:

from keras.layers import Embedding, SimpleRNN, Dense, LSTM, GRU, Bidirectional, TimeDistributed, Input
from keras_contrib.layers import CRF
from keras_contrib.losses import crf_loss
from keras_contrib.metrics import crf_accuracy 
import tensorflow as tf
from keras.optimizers import RMSprop
import keras


optimizer = Adam(learning_rate=0.0005, beta_1=0.9, beta_2=0.999, amsgrad=False)
input = Input(shape=(max_length,))
model= Embedding(vocab_size, embedding_dimension, embeddings_initializer="uniform", trainable=False)(input)
model= LSTM(360, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal())(model)
model= Bidirectional(LSTM(180, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal()))(model)
model= TimeDistributed(Dense(27, activation='relu'))(model)
crf = CRF(units=27, sparse_target=True)
out = crf(model)
model = keras.Model(input, out)
model.compile(optimizer=optimizer, loss=crf_loss, metrics=[crf_accuracy, 'accuracy'])
model.summary()

训练代码:

history = model.fit(
    train_text,train_labels,
    validation_data=(val_text, val_labels),
    epochs=10,
    batch_size=32,
    verbose=1,
    callbacks=[MacroF1Callback((val_text, val_labels), (train_text, train_labels))]
)

报错信息:

----> 7 history = model.fit(
      8     train_text,train_labels,
      9     validation_data=(val_text, val_labels),

c:\keras\src\utils\traceback_utils.py in error_handler(*args, **kwargs)
    121             # To get the full stack trace, call:
    122             # `keras.config.disable_traceback_filtering()`
--> 123             raise e.with_traceback(filtered_tb) from None
    124         finally:
    125             del filtered_tb

c:\Python310\lib\site-packages\keras_contrib\layers\crf.py in call(self, X, mask)
    290 
    291         if self.test_mode == 'viterbi':
--> 292             test_output = self.viterbi_decoding(X, mask)
    293         else:
    294             test_output = self.get_marginal_prob(X, mask)

c:\Python310\lib\site-packages\keras_contrib\layers\crf.py in viterbi_decoding(self, X, mask)
    557 
...
**module 'keras.backend' has no attribute 'dot'**

Arguments received by CRF.call():
  • X=tf.Tensor(shape=(None, 78, 27), dtype=float32)
  • mask=None
解决方案

方案1:自定义适配Keras 3的CRF层

基于TensorFlow原生API实现CRF层,完全兼容Keras 3和TensorFlow 2.16,无需依赖废弃库。

自定义CRF核心代码

import tensorflow as tf
from keras import layers, backend as K
from keras.layers import Layer
from keras.losses import Loss
from keras.metrics import Metric

def crf_log_likelihood(inputs, tags, mask=None, transition_params=None):
    """计算CRF对数似然"""
    sequence_lengths = tf.reduce_sum(tf.cast(mask, tf.int32), axis=1) if mask is not None else tf.fill(tf.shape(inputs)[0], tf.shape(inputs)[1])
    inputs = tf.cast(inputs, tf.float32)
    
    if transition_params is None:
        transition_params = tf.Variable(tf.random.normal([tf.shape(inputs)[2], tf.shape(inputs)[2]]))
    
    # 计算序列分数与转移分数
    sequence_scores = tf.reduce_sum(tf.gather_nd(inputs, tf.stack([tf.tile(tf.range(tf.shape(inputs)[0])[:, tf.newaxis], [1, tf.shape(inputs)[1]]), tags], axis=-1)), axis=1)
    transition_scores = tf.reduce_sum(tf.gather_nd(transition_params, tf.stack([tags[:, :-1], tags[:, 1:]], axis=-1)), axis=1)
    total_scores = sequence_scores + transition_scores
    
    # Forward算法计算归一化项
    def forward_step(state, inputs):
        state = tf.expand_dims(state, axis=1) + transition_params + tf.expand_dims(inputs, axis=0)
        return tf.reduce_logsumexp(state, axis=0)
    
    initial_forward = inputs[:, 0]
    forward_scores = tf.scan(forward_step, tf.transpose(inputs[:, 1:], [1, 0, 2]), initializer=initial_forward)
    final_forward = tf.reduce_logsumexp(forward_scores[-1], axis=1)
    
    log_likelihood = total_scores - final_forward
    return log_likelihood, transition_params

def crf_decode(inputs, transition_params, mask=None):
    """Viterbi解码得到最优标签序列"""
    sequence_lengths = tf.reduce_sum(tf.cast(mask, tf.int32), axis=1) if mask is not None else tf.fill(tf.shape(inputs)[0], tf.shape(inputs)[1])
    inputs = tf.cast(inputs, tf.float32)
    
    def viterbi_step(state, inputs):
        prev_viterbi, _ = state
        current_viterbi = tf.expand_dims(prev_viterbi, axis=1) + transition_params + tf.expand_dims(inputs, axis=0)
        max_viterbi = tf.reduce_max(current_viterbi, axis=0)
        max_tags = tf.argmax(current_viterbi, axis=0)
        return max_viterbi, max_tags
    
    initial_viterbi = inputs[:, 0]
    initial_tags = tf.zeros(tf.shape(inputs)[0], dtype=tf.int32)
    
    viterbi_scores, tag_indices = tf.scan(viterbi_step, tf.transpose(inputs[:, 1:], [1, 0, 2]), initializer=(initial_viterbi, initial_tags))
    
    # 回溯得到最优路径
    final_tags = tf.TensorArray(tf.int32, size=tf.shape(inputs)[1])
    final_tags = final_tags.write(0, tf.argmax(viterbi_scores[-1], axis=1))
    
    for i in tf.range(tf.shape(inputs)[1]-2, -1, -1):
        final_tags = final_tags.write(i+1, tf.gather(tag_indices[i], final_tags.read(i)))
    
    final_tags = tf.transpose(final_tags.stack(), [1, 0])
    return final_tags, tf.reduce_max(viterbi_scores[-1], axis=1)

class CRFLayer(Layer):
    def __init__(self, num_labels, sparse_target=True, **kwargs):
        super().__init__(**kwargs)
        self.num_labels = num_labels
        self.sparse_target = sparse_target
        self.transitions = self.add_weight(
            name="transitions",
            shape=(num_labels, num_labels),
            initializer="glorot_uniform",
            trainable=True,
        )
        self._targets = None

    def call(self, inputs, mask=None, training=None):
        if training:
            assert self._targets is not None, "训练时需通过损失函数传入标签"
            log_likelihood, self.transitions = crf_log_likelihood(
                inputs, self._targets, mask, transition_params=self.transitions
            )
            self.add_loss(-K.mean(log_likelihood))
            return inputs
        else:
            viterbi_sequence, _ = crf_decode(inputs, self.transitions, mask)
            return viterbi_sequence

    def set_targets(self, targets):
        self._targets = K.cast(targets, dtype=tf.int32) if self.sparse_target else K.argmax(targets, axis=-1)

    def get_config(self):
        config = super().get_config()
        config.update({"num_labels": self.num_labels, "sparse_target": self.sparse_target})
        return config

class CRFLoss(Loss):
    def __init__(self, crf_layer, name="crf_loss"):
        super().__init__(name=name)
        self.crf_layer = crf_layer

    def call(self, y_true, y_pred):
        self.crf_layer.set_targets(y_true)
        log_likelihood, _ = crf_log_likelihood(y_pred, self.crf_layer._targets, None, self.crf_layer.transitions)
        return -K.mean(log_likelihood)

class CRFAccuracy(Metric):
    def __init__(self, name="crf_accuracy", **kwargs):
        super().__init__(name=name, **kwargs)
        self.accuracy = self.add_weight(name="acc", initializer="zeros")
        self.total = self.add_weight(name="total", initializer="zeros")

    def update_state(self, y_true, y_pred, sample_weight=None):
        y_true = K.cast(y_true, tf.int32)
        y_pred = K.cast(y_pred, tf.int32)
        matches = K.cast(K.equal(y_true, y_pred), K.floatx())
        if sample_weight is not None:
            matches *= K.cast(sample_weight, K.floatx())
            self.total.assign_add(K.sum(sample_weight))
        else:
            self.total.assign_add(K.cast(K.size(y_true), K.floatx()))
        self.accuracy.assign_add(K.sum(matches))

    def result(self):
        return self.accuracy / self.total

    def reset_state(self):
        self.accuracy.assign(0.0)
        self.total.assign(0.0)

修改后的模型构建代码

from keras.layers import Embedding, LSTM, Bidirectional, TimeDistributed, Input
from keras.optimizers import Adam
import keras

optimizer = Adam(learning_rate=0.0005, beta_1=0.9, beta_2=0.999, amsgrad=False)
input = Input(shape=(max_length,))
model= Embedding(vocab_size, embedding_dimension, embeddings_initializer="uniform", trainable=False)(input)
model= LSTM(360, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal())(model)
model= Bidirectional(LSTM(180, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal()))(model)
model= TimeDistributed(Dense(27, activation='relu'))(model)

# 替换为自定义CRF层
crf_layer = CRFLayer(num_labels=27, sparse_target=True)
out = crf_layer(model)
model = keras.Model(input, out)

model.compile(optimizer=optimizer, loss=CRFLoss(crf_layer), metrics=[CRFAccuracy(), 'accuracy'])
model.summary()

方案2:使用Keras NLP官方CRF组件

Keras NLP是Keras官方维护的NLP工具库,提供了完全兼容Keras 3的CRF层,无需自定义。

步骤1:安装依赖

pip install keras-nlp

替换后的模型代码

from keras.layers import Embedding, LSTM, Bidirectional, TimeDistributed, Input
from keras.optimizers import Adam
from keras_nlp.layers import CRF
import keras

optimizer = Adam(learning_rate=0.0005, beta_1=0.9, beta_2=0.999, amsgrad=False)
input = Input(shape=(max_length,))
model= Embedding(vocab_size, embedding_dimension, embeddings_initializer="uniform", trainable=False)(input)
model= LSTM(360, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal())(model)
model= Bidirectional(LSTM(180, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal()))(model)
model= TimeDistributed(Dense(27, activation='relu'))(model)

# 使用Keras NLP官方CRF层
crf = CRF(num_labels=27, sparse_target=True)
out = crf(model)
model = keras.Model(input, out)

model.compile(optimizer=optimizer, loss=crf.loss, metrics=[crf.accuracy, 'accuracy'])
model.summary()

该层内置了适配Keras 3的损失函数和准确率指标,直接调用即可。

方案3:改用Transformer模型替代BiLSTM-CRF(推荐)

如果不严格依赖BiLSTM结构,BERT等Transformer模型在NER任务上表现更优,且Keras 3对预训练Transformer支持完善,无需处理CRF兼容性问题。

示例代码(BERT做NER)

import keras_nlp
import tensorflow as tf
from keras import layers

# 加载预训练BERT组件
preprocessor = keras_nlp.models.BertPreprocessor.from_preset("bert_base_en_uncased")
backbone = keras_nlp.models.BertBackbone.from_preset("bert_base_en_uncased")

# 构建NER模型
inputs = preprocessor.input
sequence_output = backbone(inputs)["sequence_output"]
outputs = layers.Dense(num_labels, activation="softmax")(sequence_output)
model = keras_nlp.models.TaskModel(inputs, outputs, preprocessor=preprocessor)

model.compile(
    optimizer=tf.keras.optimizers.AdamW(learning_rate=5e-5),
    loss=tf.keras.losses.SparseCategoricalCrossentropy(),
    metrics=[tf.keras.metrics.SparseCategoricalAccuracy()],
)

# 训练模型
model.fit(train_dataset, validation_data=val_dataset, epochs=3)

内容的提问来源于stack exchange,提问作者TACO TRAIN

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.28 00:49:51