TensorFlow2.16+Keras3.0下BiLSTM-CRF命名实体识别替代方案求助
问题
在TensorFlow 2.16 + Keras 3.0环境下构建BiLSTM-CRF模型完成命名实体识别(NER)任务时,使用已废弃的keras_contrib或tensorflow_addons提供的CRF层出现严重兼容性问题(如运行时报错module 'keras.backend' has no attribute 'dot'),且不愿降级TensorFlow版本,求可行的替代方案。
原实现代码:
from keras.layers import Embedding, SimpleRNN, Dense, LSTM, GRU, Bidirectional, TimeDistributed, Input from keras_contrib.layers import CRF from keras_contrib.losses import crf_loss from keras_contrib.metrics import crf_accuracy import tensorflow as tf from keras.optimizers import RMSprop import keras optimizer = Adam(learning_rate=0.0005, beta_1=0.9, beta_2=0.999, amsgrad=False) input = Input(shape=(max_length,)) model= Embedding(vocab_size, embedding_dimension, embeddings_initializer="uniform", trainable=False)(input) model= LSTM(360, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal())(model) model= Bidirectional(LSTM(180, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal()))(model) model= TimeDistributed(Dense(27, activation='relu'))(model) crf = CRF(units=27, sparse_target=True) out = crf(model) model = keras.Model(input, out) model.compile(optimizer=optimizer, loss=crf_loss, metrics=[crf_accuracy, 'accuracy']) model.summary()
训练代码:
history = model.fit( train_text,train_labels, validation_data=(val_text, val_labels), epochs=10, batch_size=32, verbose=1, callbacks=[MacroF1Callback((val_text, val_labels), (train_text, train_labels))] )
报错信息:
----> 7 history = model.fit( 8 train_text,train_labels, 9 validation_data=(val_text, val_labels), c:\keras\src\utils\traceback_utils.py in error_handler(*args, **kwargs) 121 # To get the full stack trace, call: 122 # `keras.config.disable_traceback_filtering()` --> 123 raise e.with_traceback(filtered_tb) from None 124 finally: 125 del filtered_tb c:\Python310\lib\site-packages\keras_contrib\layers\crf.py in call(self, X, mask) 290 291 if self.test_mode == 'viterbi': --> 292 test_output = self.viterbi_decoding(X, mask) 293 else: 294 test_output = self.get_marginal_prob(X, mask) c:\Python310\lib\site-packages\keras_contrib\layers\crf.py in viterbi_decoding(self, X, mask) 557 ... **module 'keras.backend' has no attribute 'dot'** Arguments received by CRF.call(): • X=tf.Tensor(shape=(None, 78, 27), dtype=float32) • mask=None
解决方案
方案1:自定义适配Keras 3的CRF层
基于TensorFlow原生API实现CRF层,完全兼容Keras 3和TensorFlow 2.16,无需依赖废弃库。
自定义CRF核心代码
import tensorflow as tf from keras import layers, backend as K from keras.layers import Layer from keras.losses import Loss from keras.metrics import Metric def crf_log_likelihood(inputs, tags, mask=None, transition_params=None): """计算CRF对数似然""" sequence_lengths = tf.reduce_sum(tf.cast(mask, tf.int32), axis=1) if mask is not None else tf.fill(tf.shape(inputs)[0], tf.shape(inputs)[1]) inputs = tf.cast(inputs, tf.float32) if transition_params is None: transition_params = tf.Variable(tf.random.normal([tf.shape(inputs)[2], tf.shape(inputs)[2]])) # 计算序列分数与转移分数 sequence_scores = tf.reduce_sum(tf.gather_nd(inputs, tf.stack([tf.tile(tf.range(tf.shape(inputs)[0])[:, tf.newaxis], [1, tf.shape(inputs)[1]]), tags], axis=-1)), axis=1) transition_scores = tf.reduce_sum(tf.gather_nd(transition_params, tf.stack([tags[:, :-1], tags[:, 1:]], axis=-1)), axis=1) total_scores = sequence_scores + transition_scores # Forward算法计算归一化项 def forward_step(state, inputs): state = tf.expand_dims(state, axis=1) + transition_params + tf.expand_dims(inputs, axis=0) return tf.reduce_logsumexp(state, axis=0) initial_forward = inputs[:, 0] forward_scores = tf.scan(forward_step, tf.transpose(inputs[:, 1:], [1, 0, 2]), initializer=initial_forward) final_forward = tf.reduce_logsumexp(forward_scores[-1], axis=1) log_likelihood = total_scores - final_forward return log_likelihood, transition_params def crf_decode(inputs, transition_params, mask=None): """Viterbi解码得到最优标签序列""" sequence_lengths = tf.reduce_sum(tf.cast(mask, tf.int32), axis=1) if mask is not None else tf.fill(tf.shape(inputs)[0], tf.shape(inputs)[1]) inputs = tf.cast(inputs, tf.float32) def viterbi_step(state, inputs): prev_viterbi, _ = state current_viterbi = tf.expand_dims(prev_viterbi, axis=1) + transition_params + tf.expand_dims(inputs, axis=0) max_viterbi = tf.reduce_max(current_viterbi, axis=0) max_tags = tf.argmax(current_viterbi, axis=0) return max_viterbi, max_tags initial_viterbi = inputs[:, 0] initial_tags = tf.zeros(tf.shape(inputs)[0], dtype=tf.int32) viterbi_scores, tag_indices = tf.scan(viterbi_step, tf.transpose(inputs[:, 1:], [1, 0, 2]), initializer=(initial_viterbi, initial_tags)) # 回溯得到最优路径 final_tags = tf.TensorArray(tf.int32, size=tf.shape(inputs)[1]) final_tags = final_tags.write(0, tf.argmax(viterbi_scores[-1], axis=1)) for i in tf.range(tf.shape(inputs)[1]-2, -1, -1): final_tags = final_tags.write(i+1, tf.gather(tag_indices[i], final_tags.read(i))) final_tags = tf.transpose(final_tags.stack(), [1, 0]) return final_tags, tf.reduce_max(viterbi_scores[-1], axis=1) class CRFLayer(Layer): def __init__(self, num_labels, sparse_target=True, **kwargs): super().__init__(**kwargs) self.num_labels = num_labels self.sparse_target = sparse_target self.transitions = self.add_weight( name="transitions", shape=(num_labels, num_labels), initializer="glorot_uniform", trainable=True, ) self._targets = None def call(self, inputs, mask=None, training=None): if training: assert self._targets is not None, "训练时需通过损失函数传入标签" log_likelihood, self.transitions = crf_log_likelihood( inputs, self._targets, mask, transition_params=self.transitions ) self.add_loss(-K.mean(log_likelihood)) return inputs else: viterbi_sequence, _ = crf_decode(inputs, self.transitions, mask) return viterbi_sequence def set_targets(self, targets): self._targets = K.cast(targets, dtype=tf.int32) if self.sparse_target else K.argmax(targets, axis=-1) def get_config(self): config = super().get_config() config.update({"num_labels": self.num_labels, "sparse_target": self.sparse_target}) return config class CRFLoss(Loss): def __init__(self, crf_layer, name="crf_loss"): super().__init__(name=name) self.crf_layer = crf_layer def call(self, y_true, y_pred): self.crf_layer.set_targets(y_true) log_likelihood, _ = crf_log_likelihood(y_pred, self.crf_layer._targets, None, self.crf_layer.transitions) return -K.mean(log_likelihood) class CRFAccuracy(Metric): def __init__(self, name="crf_accuracy", **kwargs): super().__init__(name=name, **kwargs) self.accuracy = self.add_weight(name="acc", initializer="zeros") self.total = self.add_weight(name="total", initializer="zeros") def update_state(self, y_true, y_pred, sample_weight=None): y_true = K.cast(y_true, tf.int32) y_pred = K.cast(y_pred, tf.int32) matches = K.cast(K.equal(y_true, y_pred), K.floatx()) if sample_weight is not None: matches *= K.cast(sample_weight, K.floatx()) self.total.assign_add(K.sum(sample_weight)) else: self.total.assign_add(K.cast(K.size(y_true), K.floatx())) self.accuracy.assign_add(K.sum(matches)) def result(self): return self.accuracy / self.total def reset_state(self): self.accuracy.assign(0.0) self.total.assign(0.0)
修改后的模型构建代码
from keras.layers import Embedding, LSTM, Bidirectional, TimeDistributed, Input from keras.optimizers import Adam import keras optimizer = Adam(learning_rate=0.0005, beta_1=0.9, beta_2=0.999, amsgrad=False) input = Input(shape=(max_length,)) model= Embedding(vocab_size, embedding_dimension, embeddings_initializer="uniform", trainable=False)(input) model= LSTM(360, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal())(model) model= Bidirectional(LSTM(180, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal()))(model) model= TimeDistributed(Dense(27, activation='relu'))(model) # 替换为自定义CRF层 crf_layer = CRFLayer(num_labels=27, sparse_target=True) out = crf_layer(model) model = keras.Model(input, out) model.compile(optimizer=optimizer, loss=CRFLoss(crf_layer), metrics=[CRFAccuracy(), 'accuracy']) model.summary()
方案2:使用Keras NLP官方CRF组件
Keras NLP是Keras官方维护的NLP工具库,提供了完全兼容Keras 3的CRF层,无需自定义。
步骤1:安装依赖
pip install keras-nlp
替换后的模型代码
from keras.layers import Embedding, LSTM, Bidirectional, TimeDistributed, Input from keras.optimizers import Adam from keras_nlp.layers import CRF import keras optimizer = Adam(learning_rate=0.0005, beta_1=0.9, beta_2=0.999, amsgrad=False) input = Input(shape=(max_length,)) model= Embedding(vocab_size, embedding_dimension, embeddings_initializer="uniform", trainable=False)(input) model= LSTM(360, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal())(model) model= Bidirectional(LSTM(180, return_sequences=True, dropout=0.2, recurrent_dropout=0.2, kernel_initializer=keras.initializers.he_normal()))(model) model= TimeDistributed(Dense(27, activation='relu'))(model) # 使用Keras NLP官方CRF层 crf = CRF(num_labels=27, sparse_target=True) out = crf(model) model = keras.Model(input, out) model.compile(optimizer=optimizer, loss=crf.loss, metrics=[crf.accuracy, 'accuracy']) model.summary()
该层内置了适配Keras 3的损失函数和准确率指标,直接调用即可。
方案3:改用Transformer模型替代BiLSTM-CRF(推荐)
如果不严格依赖BiLSTM结构,BERT等Transformer模型在NER任务上表现更优,且Keras 3对预训练Transformer支持完善,无需处理CRF兼容性问题。
示例代码(BERT做NER)
import keras_nlp import tensorflow as tf from keras import layers # 加载预训练BERT组件 preprocessor = keras_nlp.models.BertPreprocessor.from_preset("bert_base_en_uncased") backbone = keras_nlp.models.BertBackbone.from_preset("bert_base_en_uncased") # 构建NER模型 inputs = preprocessor.input sequence_output = backbone(inputs)["sequence_output"] outputs = layers.Dense(num_labels, activation="softmax")(sequence_output) model = keras_nlp.models.TaskModel(inputs, outputs, preprocessor=preprocessor) model.compile( optimizer=tf.keras.optimizers.AdamW(learning_rate=5e-5), loss=tf.keras.losses.SparseCategoricalCrossentropy(), metrics=[tf.keras.metrics.SparseCategoricalAccuracy()], ) # 训练模型 model.fit(train_dataset, validation_data=val_dataset, epochs=3)
内容的提问来源于stack exchange,提问作者TACO TRAIN
相关产品推荐
相关产品推荐

