加载.keras格式NER模型遇TypeError:NERModel.__init__()获意外参数'trainable'
解决Keras自定义NER模型加载失败问题
问题现象
训练自定义命名实体识别(NER)模型并保存为.keras格式后,加载时触发如下错误:
TypeError: <class 'modeling.NERModel'> could not be deserialized properly. Please ensure that components that are Python object instances (layers, models, etc.) returned by `get_config()` are explicitly deserialized in the model's `from_config()` method. config={'module': 'modeling', 'class_name': 'NERModel', 'config': {'trainable': True, 'dtype': 'float32'}, 'registered_name': 'NERModel', 'build_config': {'input_shape': [None, None]}, 'compile_config': {'optimizer': 'adam', 'loss': {'module': 'modeling', 'class_name': 'CustomNonPaddingTokenLoss', 'config': {'name': 'custom_ner_loss', 'reduction': 'sum'}, 'registered_name': 'CustomNonPaddingTokenLoss'}, 'loss_weights': None, 'metrics': None, 'weighted_metrics': None, 'run_eagerly': False, 'steps_per_execution': 1, 'jit_compile': False}}. Exception encountered: Unable to revive model from config. When overriding the `get_config()` method, make sure that the returned config contains all items used as arguments in the constructor to <class 'modeling.NERModel'>, which is the default behavior. You can override this default behavior by defining a `from_config(cls, config)` class method to specify how to create an instance of NERModel from its config. Received config={'trainable': True, 'dtype': 'float32'} Error encountered during deserialization: NERModel.__init__() got an unexpected keyword argument 'trainable'
训练时的模型代码
import os from tensorflow import keras import tensorflow as tf from keras import layers class TransformerBlock(layers.Layer): def __init__(self, embed_dim, num_heads, ff_dim, rate=0.1): super().__init__() self.att = keras.layers.MultiHeadAttention( num_heads=num_heads, key_dim=embed_dim ) self.ffn = keras.Sequential( [ keras.layers.Dense(ff_dim, activation="relu"), keras.layers.Dense(embed_dim), ] ) self.layernorm1 = keras.layers.LayerNormalization(epsilon=1e-6) self.layernorm2 = keras.layers.LayerNormalization(epsilon=1e-6) self.dropout1 = keras.layers.Dropout(rate) self.dropout2 = keras.layers.Dropout(rate) def call(self, inputs, training=False): attn_output = self.att(inputs, inputs) attn_output = self.dropout1(attn_output, training=training) out1 = self.layernorm1(inputs + attn_output) ffn_output = self.ffn(out1) ffn_output = self.dropout2(ffn_output, training=training) return self.layernorm2(out1 + ffn_output) class TokenAndPositionEmbedding(layers.Layer): def __init__(self, maxlen, vocab_size, embed_dim): super().__init__() self.token_emb = keras.layers.Embedding( input_dim=vocab_size, output_dim=embed_dim ) self.pos_emb = keras.layers.Embedding(input_dim=maxlen, output_dim=embed_dim) def call(self, inputs): maxlen = tf.shape(inputs)[-1] positions = tf.range(start=0, limit=maxlen, delta=1) position_embeddings = self.pos_emb(positions) token_embeddings = self.token_emb(inputs) return token_embeddings + position_embeddings class NERModel(keras.Model): def __init__( self, num_tags, vocab_size, maxlen=1000, embed_dim=32, num_heads=2, ff_dim=32 ): super().__init__() self.embedding_layer = TokenAndPositionEmbedding(maxlen, vocab_size, embed_dim) self.transformer_block = TransformerBlock(embed_dim, num_heads, ff_dim) self.dropout1 = layers.Dropout(0.1) self.ff = layers.Dense(ff_dim, activation="relu") self.dropout2 = layers.Dropout(0.1) self.ff_final = layers.Dense(num_tags, activation="softmax") def call(self, inputs, training=False): x = self.embedding_layer(inputs) x = self.transformer_block(x) x = self.dropout1(x, training=training) x = self.ff(x) x = self.dropout2(x, training=training) x = self.ff_final(x) return x class CustomNonPaddingTokenLoss(keras.losses.Loss): def __init__(self, reduction='sum', name="custom_ner_loss"): super().__init__(reduction=reduction, name=name) def call(self, y_true, y_pred): loss_fn = keras.losses.SparseCategoricalCrossentropy( from_logits=False, reduction=self.reduction # Pass the reduction argument here ) loss = loss_fn(y_true, y_pred) mask = tf.cast((y_true > 0), dtype=tf.float32) loss = loss * mask return tf.reduce_sum(loss) / tf.reduce_sum(mask) def save_model(model, filepath): if os.path.exists(filepath): filepath = filepath[:-6] + "1" + filepath[-6:] model.save(filepath)
预测加载代码
import re import pickle import keras import tensorflow as tf import numpy as np from data_preprocess import map_record_to_training_data from modeling import CustomNonPaddingTokenLoss def lookup(tokens): # Load the list from the file with open('./resources/vocabulary.pkl', 'rb') as f: loaded_list = pickle.load(f) # The StringLookup class will convert tokens to token IDs lookup_layer = keras.layers.StringLookup(vocabulary=loaded_list) return lookup_layer(tokens) def format_datatype(data): tokens = [re.sub(r'[;,]', '', d) for d in data.split(' ')] #default is 0, since is for prediction ner_tags = [0 for d in data.split(' ')] #tab to separate string_input = str(len(tokens))+ "\t"+ "\t".join(tokens)+ "\t"+ "\t".join(map(str, ner_tags)) string_input = tf.data.Dataset.from_tensor_slices([string_input]) finalize_input = (string_input.map(map_record_to_training_data) .map(lambda x, y: (lookup(x), y)) .padded_batch(1) ) return finalize_input def prediction(data): # Load model # Load the model loaded_model = tf.keras.models.load_model("./resources/trained_model/ner_model.keras", custom_objects={'CustomNonPaddingTokenLoss': CustomNonPaddingTokenLoss}) print(loaded_model.summary()) all_predicted_tag_ids = [] for x, _ in data: print("Input Tensor Info:") print("Data Type:", x.dtype) print("Shape:", x.shape) output = loaded_model(x, training=False) predictions = np.argmax(output, axis=-1) predictions = np.reshape(predictions, [-1]) all_predicted_tag_ids.append(predictions) all_predicted_tag_ids = np.concatenate(all_predicted_tag_ids) ner_labels = ["[PAD]", "N", "M", "other"] mapping = dict(zip(range(len(ner_labels)), ner_labels)) predicted_tags = [mapping[tag] for tag in all_predicted_tag_ids] return predicted_tags sample_input = "Hi, my name is David" result = prediction(format_datatype(sample_input)) print(result) print(sample_input.split(' ')) print(len(result))
问题原因
- 自定义模型未保存初始化参数:
NERModel继承自keras.Model,默认的get_config()仅返回父类的通用配置(如trainable、dtype),但模型初始化需要的核心参数(num_tags、vocab_size等)未被保存,导致加载时无法重建模型实例。 - 加载时未指定所有自定义组件:加载模型时仅传入了
CustomNonPaddingTokenLoss,但自定义的NERModel、TransformerBlock、TokenAndPositionEmbedding也需要被Keras识别才能完成反序列化。 - 参数不匹配:Keras加载时会将默认config中的
trainable等参数传入模型构造函数,但NERModel.__init__未定义这些参数,导致报错。
解决方案
步骤1:修改NERModel类,添加get_config()方法
重写get_config(),保存模型初始化所需的所有参数,同时兼容父类的配置:
class NERModel(keras.Model): def __init__( self, num_tags, vocab_size, maxlen=1000, embed_dim=32, num_heads=2, ff_dim=32 ): super().__init__() # 先保存初始化参数为实例属性 self.num_tags = num_tags self.vocab_size = vocab_size self.maxlen = maxlen self.embed_dim = embed_dim self.num_heads = num_heads self.ff_dim = ff_dim # 原有初始化逻辑不变 self.embedding_layer = TokenAndPositionEmbedding(maxlen, vocab_size, embed_dim) self.transformer_block = TransformerBlock(embed_dim, num_heads, ff_dim) self.dropout1 = layers.Dropout(0.1) self.ff = layers.Dense(ff_dim, activation="relu") self.dropout2 = layers.Dropout(0.1) self.ff_final = layers.Dense(num_tags, activation="softmax") def call(self, inputs, training=False): x = self.embedding_layer(inputs) x = self.transformer_block(x) x = self.dropout1(x, training=training) x = self.ff(x) x = self.dropout2(x, training=training) x = self.ff_final(x) return x def get_config(self): # 获取父类的配置 base_config = super().get_config() # 添加自定义初始化参数 custom_config = { "num_tags": self.num_tags, "vocab_size": self.vocab_size, "maxlen": self.maxlen, "embed_dim": self.embed_dim, "num_heads": self.num_heads, "ff_dim": self.ff_dim, } # 合并配置 base_config.update(custom_config) return base_config
步骤2:修改模型加载代码,添加所有自定义组件
加载模型时,将所有自定义的层和模型都加入custom_objects:
# 导入所有自定义组件 from modeling import ( CustomNonPaddingTokenLoss, NERModel, TransformerBlock, TokenAndPositionEmbedding ) def prediction(data): # 加载模型时指定所有自定义对象 loaded_model = tf.keras.models.load_model( "./resources/trained_model/ner_model.keras", custom_objects={ 'CustomNonPaddingTokenLoss': CustomNonPaddingTokenLoss, 'NERModel': NERModel, 'TransformerBlock': TransformerBlock, 'TokenAndPositionEmbedding': TokenAndPositionEmbedding } ) # 后续预测逻辑不变 print(loaded_model.summary()) all_predicted_tag_ids = [] for x, _ in data: print("Input Tensor Info:") print("Data Type:", x.dtype) print("Shape:", x.shape) output = loaded_model(x, training=False) predictions = np.argmax(output, axis=-1) predictions = np.reshape(predictions, [-1]) all_predicted_tag_ids.append(predictions) all_predicted_tag_ids = np.concatenate(all_predicted_tag_ids) ner_labels = ["[PAD]", "N", "M", "other"] mapping = dict(zip(range(len(ner_labels)), ner_labels)) predicted_tags = [mapping[tag] for tag in all_predicted_tag_ids] return predicted_tags
步骤3:重新训练并保存模型
使用修改后的代码重新训练模型,保存新的.keras文件,确保配置信息完整。
原理说明
get_config()方法负责将模型的关键参数序列化为字典,保存到模型文件中;加载时Keras会用这个字典调用模型的构造函数重建实例。custom_objects参数告诉Keras如何识别自定义的层、模型和损失函数,避免反序列化时出现类找不到的错误。
内容的提问来源于stack exchange,提问作者Yin Jie
相关产品推荐
相关产品推荐

