You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

加载.keras格式NER模型遇TypeError:NERModel.__init__()获意外参数'trainable'

解决Keras自定义NER模型加载失败问题

问题现象

训练自定义命名实体识别(NER)模型并保存为.keras格式后,加载时触发如下错误:

TypeError: <class 'modeling.NERModel'> could not be deserialized properly. Please ensure that components that are Python object instances (layers, models, etc.) returned 
by `get_config()` are explicitly deserialized in the model's `from_config()` method.

config={'module': 'modeling', 'class_name': 'NERModel', 'config': {'trainable': True, 'dtype': 'float32'}, 'registered_name': 'NERModel', 'build_config': {'input_shape': 
[None, None]}, 'compile_config': {'optimizer': 'adam', 'loss': {'module': 'modeling', 'class_name': 'CustomNonPaddingTokenLoss', 'config': {'name': 'custom_ner_loss', 'reduction': 'sum'}, 'registered_name': 'CustomNonPaddingTokenLoss'}, 'loss_weights': None, 'metrics': None, 'weighted_metrics': None, 'run_eagerly': False, 'steps_per_execution': 1, 'jit_compile': False}}.

Exception encountered: Unable to revive model from config. When overriding the `get_config()` method, make sure that the returned config contains all items used as arguments in the  constructor to <class 'modeling.NERModel'>, which is the default behavior. You can override this default behavior by defining a `from_config(cls, config)` class method to specify how to create an instance of NERModel from its config.

Received config={'trainable': True, 'dtype': 'float32'}

Error encountered during deserialization: NERModel.__init__() got an unexpected keyword argument 'trainable'

训练时的模型代码

import os
from tensorflow import keras
import tensorflow as tf
from keras import layers

class TransformerBlock(layers.Layer):
    def __init__(self, embed_dim, num_heads, ff_dim, rate=0.1):
        super().__init__()
        self.att = keras.layers.MultiHeadAttention(
            num_heads=num_heads, key_dim=embed_dim
        )
        self.ffn = keras.Sequential(
            [
                keras.layers.Dense(ff_dim, activation="relu"),
                keras.layers.Dense(embed_dim),
            ]
        )
        self.layernorm1 = keras.layers.LayerNormalization(epsilon=1e-6)
        self.layernorm2 = keras.layers.LayerNormalization(epsilon=1e-6)
        self.dropout1 = keras.layers.Dropout(rate)
        self.dropout2 = keras.layers.Dropout(rate)

    def call(self, inputs, training=False):
        attn_output = self.att(inputs, inputs)
        attn_output = self.dropout1(attn_output, training=training)
        out1 = self.layernorm1(inputs + attn_output)
        ffn_output = self.ffn(out1)
        ffn_output = self.dropout2(ffn_output, training=training)
        return self.layernorm2(out1 + ffn_output)

class TokenAndPositionEmbedding(layers.Layer):
    def __init__(self, maxlen, vocab_size, embed_dim):
        super().__init__()
        self.token_emb = keras.layers.Embedding(
            input_dim=vocab_size, output_dim=embed_dim
        )
        self.pos_emb = keras.layers.Embedding(input_dim=maxlen, output_dim=embed_dim)

    def call(self, inputs):
        maxlen = tf.shape(inputs)[-1]
        positions = tf.range(start=0, limit=maxlen, delta=1)
        position_embeddings = self.pos_emb(positions)
        token_embeddings = self.token_emb(inputs)
        return token_embeddings + position_embeddings
    
class NERModel(keras.Model):
    def __init__(
        self, num_tags, vocab_size, maxlen=1000, embed_dim=32, num_heads=2, ff_dim=32
    ):
        super().__init__()
        self.embedding_layer = TokenAndPositionEmbedding(maxlen, vocab_size, embed_dim)
        self.transformer_block = TransformerBlock(embed_dim, num_heads, ff_dim)
        self.dropout1 = layers.Dropout(0.1)
        self.ff = layers.Dense(ff_dim, activation="relu")
        self.dropout2 = layers.Dropout(0.1)
        self.ff_final = layers.Dense(num_tags, activation="softmax")

    def call(self, inputs, training=False):
        x = self.embedding_layer(inputs)
        x = self.transformer_block(x)
        x = self.dropout1(x, training=training)
        x = self.ff(x)
        x = self.dropout2(x, training=training)
        x = self.ff_final(x)
        return x

class CustomNonPaddingTokenLoss(keras.losses.Loss):
    def __init__(self, reduction='sum', name="custom_ner_loss"):
        super().__init__(reduction=reduction, name=name)

    def call(self, y_true, y_pred):
        loss_fn = keras.losses.SparseCategoricalCrossentropy(
            from_logits=False, reduction=self.reduction  # Pass the reduction argument here
        )
        loss = loss_fn(y_true, y_pred)
        mask = tf.cast((y_true > 0), dtype=tf.float32)
        loss = loss * mask
        return tf.reduce_sum(loss) / tf.reduce_sum(mask)
    
def save_model(model, filepath):
    if os.path.exists(filepath):
        filepath = filepath[:-6] + "1" + filepath[-6:]
    model.save(filepath)

预测加载代码

import re
import pickle
import keras
import tensorflow as tf
import numpy as np
from data_preprocess import map_record_to_training_data
from modeling import CustomNonPaddingTokenLoss
    
def lookup(tokens):
    # Load the list from the file
    with open('./resources/vocabulary.pkl', 'rb') as f:
        loaded_list = pickle.load(f)
    # The StringLookup class will convert tokens to token IDs
    lookup_layer = keras.layers.StringLookup(vocabulary=loaded_list)

    return lookup_layer(tokens)

def format_datatype(data):
    tokens =  [re.sub(r'[;,]', '', d) for d in data.split(' ')]
    #default is 0, since is for prediction
    ner_tags = [0 for d in data.split(' ')]

    #tab to separate
    string_input = str(len(tokens))+ "\t"+ "\t".join(tokens)+ "\t"+ "\t".join(map(str, ner_tags))
    string_input = tf.data.Dataset.from_tensor_slices([string_input])


    finalize_input = (string_input.map(map_record_to_training_data)
                      .map(lambda x, y: (lookup(x),  y))
                      .padded_batch(1)
                      )

    return finalize_input

def prediction(data):
    # Load model
    # Load the model
    loaded_model = tf.keras.models.load_model("./resources/trained_model/ner_model.keras", 
                                        custom_objects={'CustomNonPaddingTokenLoss': CustomNonPaddingTokenLoss})

    print(loaded_model.summary())
    all_predicted_tag_ids = []

    for x, _ in data:
        print("Input Tensor Info:")
        print("Data Type:", x.dtype)
        print("Shape:", x.shape)
        output = loaded_model(x, training=False)
        predictions = np.argmax(output, axis=-1)
        predictions = np.reshape(predictions, [-1])
        all_predicted_tag_ids.append(predictions)

    all_predicted_tag_ids = np.concatenate(all_predicted_tag_ids)

    ner_labels = ["[PAD]", "N", "M", "other"]
    mapping =  dict(zip(range(len(ner_labels)), ner_labels))
    predicted_tags = [mapping[tag] for tag in all_predicted_tag_ids]

    return predicted_tags


sample_input = "Hi, my name is David"
result = prediction(format_datatype(sample_input))
print(result)
print(sample_input.split(' '))
print(len(result))

问题原因

  1. 自定义模型未保存初始化参数:NERModel继承自keras.Model,默认的get_config()仅返回父类的通用配置(如trainable、dtype),但模型初始化需要的核心参数(num_tags、vocab_size等)未被保存,导致加载时无法重建模型实例。
  2. 加载时未指定所有自定义组件:加载模型时仅传入了CustomNonPaddingTokenLoss,但自定义的NERModel、TransformerBlock、TokenAndPositionEmbedding也需要被Keras识别才能完成反序列化。
  3. 参数不匹配:Keras加载时会将默认config中的trainable等参数传入模型构造函数,但NERModel.__init__未定义这些参数,导致报错。

解决方案

步骤1:修改NERModel类,添加get_config()方法

重写get_config(),保存模型初始化所需的所有参数,同时兼容父类的配置:

class NERModel(keras.Model):
    def __init__(
        self, num_tags, vocab_size, maxlen=1000, embed_dim=32, num_heads=2, ff_dim=32
    ):
        super().__init__()
        # 先保存初始化参数为实例属性
        self.num_tags = num_tags
        self.vocab_size = vocab_size
        self.maxlen = maxlen
        self.embed_dim = embed_dim
        self.num_heads = num_heads
        self.ff_dim = ff_dim
        # 原有初始化逻辑不变
        self.embedding_layer = TokenAndPositionEmbedding(maxlen, vocab_size, embed_dim)
        self.transformer_block = TransformerBlock(embed_dim, num_heads, ff_dim)
        self.dropout1 = layers.Dropout(0.1)
        self.ff = layers.Dense(ff_dim, activation="relu")
        self.dropout2 = layers.Dropout(0.1)
        self.ff_final = layers.Dense(num_tags, activation="softmax")

    def call(self, inputs, training=False):
        x = self.embedding_layer(inputs)
        x = self.transformer_block(x)
        x = self.dropout1(x, training=training)
        x = self.ff(x)
        x = self.dropout2(x, training=training)
        x = self.ff_final(x)
        return x

    def get_config(self):
        # 获取父类的配置
        base_config = super().get_config()
        # 添加自定义初始化参数
        custom_config = {
            "num_tags": self.num_tags,
            "vocab_size": self.vocab_size,
            "maxlen": self.maxlen,
            "embed_dim": self.embed_dim,
            "num_heads": self.num_heads,
            "ff_dim": self.ff_dim,
        }
        # 合并配置
        base_config.update(custom_config)
        return base_config

步骤2:修改模型加载代码,添加所有自定义组件

加载模型时,将所有自定义的层和模型都加入custom_objects:

# 导入所有自定义组件
from modeling import (
    CustomNonPaddingTokenLoss, 
    NERModel, 
    TransformerBlock, 
    TokenAndPositionEmbedding
)

def prediction(data):
    # 加载模型时指定所有自定义对象
    loaded_model = tf.keras.models.load_model(
        "./resources/trained_model/ner_model.keras", 
        custom_objects={
            'CustomNonPaddingTokenLoss': CustomNonPaddingTokenLoss,
            'NERModel': NERModel,
            'TransformerBlock': TransformerBlock,
            'TokenAndPositionEmbedding': TokenAndPositionEmbedding
        }
    )

    # 后续预测逻辑不变
    print(loaded_model.summary())
    all_predicted_tag_ids = []

    for x, _ in data:
        print("Input Tensor Info:")
        print("Data Type:", x.dtype)
        print("Shape:", x.shape)
        output = loaded_model(x, training=False)
        predictions = np.argmax(output, axis=-1)
        predictions = np.reshape(predictions, [-1])
        all_predicted_tag_ids.append(predictions)

    all_predicted_tag_ids = np.concatenate(all_predicted_tag_ids)

    ner_labels = ["[PAD]", "N", "M", "other"]
    mapping =  dict(zip(range(len(ner_labels)), ner_labels))
    predicted_tags = [mapping[tag] for tag in all_predicted_tag_ids]

    return predicted_tags

步骤3:重新训练并保存模型

使用修改后的代码重新训练模型,保存新的.keras文件,确保配置信息完整。

原理说明

  • get_config()方法负责将模型的关键参数序列化为字典,保存到模型文件中;加载时Keras会用这个字典调用模型的构造函数重建实例。
  • custom_objects参数告诉Keras如何识别自定义的层、模型和损失函数,避免反序列化时出现类找不到的错误。

内容的提问来源于stack exchange,提问作者Yin Jie

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.26 06:55:54