You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Keras LSTM模型在NER多分类任务中预测标签一致的问题排查

问题:CoNLL2003 NER任务中LSTM模型始终预测相同标签

我用CoNLL2003数据集做输入,采用Word2Vec作为嵌入层构建Keras LSTM模型开展NER多分类任务,但模型一直预测相同的NER标签。尝试添加全连接层、调整epoch数量等优化操作后,问题仍未解决。以下为完整实现代码及预测结果示例:

import gensim.downloader
import pandas as pd
import numpy as np
import tensorflow as tf
from sklearn.metrics import f1_score
import string

# Load the Word2Vec model
w2v = gensim.downloader.load('word2vec-google-news-300')

# Load and preprocess the data
def load_and_preprocess_data(file_path):
    with open(file_path, 'r', encoding='utf-8') as file:
        conll_data = file.read().splitlines()
    
    # Clean rows with empty lines
    conll_data = [line for line in conll_data if line.strip() != '']
    
    # Split each line into columns
    conll_data = [line.split() for line in conll_data]
        # Remove punctuation from each word
    for i in range(len(conll_data)):
        conll_data[i] = [word.strip(string.punctuation) for word in conll_data[i]]
    
    return conll_data

train_data = load_and_preprocess_data('data/eng.train')
test_data = load_and_preprocess_data('data/eng.testb')
dev_data = load_and_preprocess_data('data/eng.testa')

# Create DataFrames
columns = ['word', 'POS', 'NP', 'label']
train_df = pd.DataFrame(train_data, columns=columns)
test_df = pd.DataFrame(test_data, columns=columns)
dev_df = pd.DataFrame(dev_data, columns=columns)

# Create labels for NER
label_to_id = {label: i for i, label in enumerate(train_df['label'].unique())}

for index, row in train_df.iterrows():
    train_df.at[index, 'label_id'] = label_to_id[row['label']]
for index, row in test_df.iterrows():
    test_df.at[index, 'label_id'] = label_to_id[row['label']]
for index, row in dev_df.iterrows():
    dev_df.at[index, 'label_id'] = label_to_id[row['label']]

texts = train_df['word'].tolist()  # list of text samples
labels_index = label_to_id  # dictionary mapping label name to numeric id
labels = train_df['label_id'].tolist()  # list of label ids
print(labels)

validation_texts = dev_df['word'].tolist()  # list of text samples
labels_index = label_to_id  # dictionary mapping label name to numeric id

# Step 1: Create a label mapping dictionary based on your training labels
label_mapping = {}

# Iterate over the unique labels in your training dataset
for label in train_df['label'].unique():
    # Map the original label to a consistent label
    label_mapping[label] = label

# Step 2: Replace validation labels with consistent labels
dev_labels = [label_mapping[label] for label in dev_df['label']]
dev_df['labels'] = dev_labels
for index, row in dev_df.iterrows():
    dev_df.at[index, 'label_id'] = label_to_id[row['label']]
validation_labels = dev_df['label_id'].tolist() 

print(len(dev_df['label'].unique()))

test_texts = test_df['word'].tolist()  # list of text samples
labels_index = label_to_id  # dictionary mapping label name to numeric id
test_labels = test_df['label_id'].tolist()  # list of label ids

from keras.preprocessing.text import Tokenizer
from keras.preprocessing.sequence import pad_sequences
from keras.utils import to_categorical
import numpy as np

# Define MAX_NB_WORDS and MAX_SEQUENCE_LENGTH before using them
MAX_NB_WORDS = 10000  # Define your desired maximum number of words
MAX_SEQUENCE_LENGTH = 100  # Define your desired maximum sequence length

# Tokenize your text data
tokenizer = Tokenizer(num_words=MAX_NB_WORDS)
tokenizer.fit_on_texts(texts)
sequences = tokenizer.texts_to_sequences(texts)

word_index = tokenizer.word_index
print('Found %s unique tokens.' % len(word_index))

data = pad_sequences(sequences, maxlen=MAX_SEQUENCE_LENGTH, padding='post', truncating='post')

labels = to_categorical(np.asarray(labels))
print('Shape of data tensor:', data.shape)
print('Shape of label tensor:', labels.shape)

# Assuming you have separate validation and test datasets
x_val = pad_sequences(tokenizer.texts_to_sequences(validation_texts), maxlen=MAX_SEQUENCE_LENGTH, padding='post', truncating='post')
y_val = padded_validation_labels = to_categorical(validation_labels, num_classes=8)
print(y_val)
print('Shape of data tensor:', x_val.shape)
print('Shape of label tensor:', y_val.shape)

x_test = pad_sequences(tokenizer.texts_to_sequences(test_texts), maxlen=MAX_SEQUENCE_LENGTH, padding='post', truncating='post')
y_test = to_categorical(np.asarray(test_labels))
print('Shape of data tensor:', x_test.shape)
print('Shape of label tensor:', y_test.shape)
to_categorical(np.asarray(test_labels))

embeddings_index = {}

# Iterate through words in the model's vocabulary and their corresponding vectors
for index, word in enumerate(w2v.index_to_key):
    embeddings_index[word] = w2v[word]

print('Found %s word vectors.' % len(embeddings_index))

# print vector for the word 'the'
print(embeddings_index['the'])

EMBEDDING_DIM = 300  # Adjust this to match the dimension of your Word2Vec model

embedding_matrix = np.zeros((len(word_index) + 1, EMBEDDING_DIM))
for word, i in word_index.items():
    embedding_vector = embeddings_index.get(word)
    if embedding_vector is not None:
        # Words not found in embedding index will be all-zeros.
        embedding_matrix[i] = embedding_vector

from keras.layers import Embedding

embedding_layer = Embedding(len(word_index) + 1,
                            EMBEDDING_DIM,
                            weights=[embedding_matrix],
                            input_length=MAX_SEQUENCE_LENGTH,
                            trainable=False)


from sklearn.metrics import f1_score
import string
from keras.preprocessing.text import Tokenizer
from keras.preprocessing.sequence import pad_sequences
from keras.utils import to_categorical
from keras.models import Sequential
from keras.layers import Embedding, LSTM, Dense
from keras.metrics import Precision, Recall, F1Score

# Create an LSTM model
model = Sequential()

# Add an Embedding layer with pre-trained Word2Vec weights
model.add(Embedding(len(word_index) + 1, EMBEDDING_DIM, weights=[embedding_matrix], input_length=MAX_SEQUENCE_LENGTH, trainable=False))
model.add(Dense(8, activation='softmax'))
# Add an LSTM layer with a specified number of units (you can adjust this)
LSTM_UNITS = 64
model.add(LSTM(LSTM_UNITS, dropout=0.2, recurrent_dropout=0.2))
# 
# Add a Dense output layer with the number of labels as units and 'softmax' activation
model.add(Dense(len(labels_index), activation='softmax'))


model.compile(
    loss='categorical_crossentropy',
    optimizer='adam',
    metrics=['accuracy', F1Score(average='micro')]
)


# Train the model
EPOCHS = 10  # You can adjust this
BATCH_SIZE = 10  # You can adjust this
history = model.fit(data, labels, epochs=EPOCHS, batch_size=BATCH_SIZE, validation_data=(x_val, y_val))

# Evaluate the model on the test data
scores = model.evaluate(x_test, y_test, verbose=0)
test_f1 = scores[model.metrics_names.index('f1_score')]
print("Test F1 Score:", test_f1)


import pandas as pd
import numpy as np

# Reverse the word and label index dictionaries
index_to_word = {v: k for k, v in word_index.items()}
index_to_labels = {v: k for k, v in label_to_id.items()}  # Replace label_to_id with your mapping

# Initialize lists to store original texts and predicted labels
# original_texts = [index_to_word.get(i, '?') for i in x_test.flatten()]  # Flatten if x_test is 2D
predicted_labels_list = []

# Make predictions
predictions = model.predict(x_test)
predicted_indices = np.argmax(predictions, axis=-1).flatten()  # Flatten if predictions is 2D

# Translate numerical indices back to original labels
predicted_labels = []
for i in predicted_indices:
    label = index_to_labels.get(i, '?')
    predicted_labels.append(label)
    
# Convert lists into DataFrame
df = pd.DataFrame({
    'Original_Texts': test_texts,
    'true labels id': test_labels,
    'true labels': test_df['label'].tolist(),
    'Predicted Labels id': predicted_indices,
    'Predicted Labels': predicted_labels
})

df.head(30)

预测结果示例

Original_Textstrue labels idtrue labelsPredicted Labels idPredicted Labels
SOCCER1.0O1O
1.0O1O
JAPAN4.0I-LOC1O
GET1.0O1O
LUCKY1.0O1O
WIN1.0O1O
1.0O1O
CHINA3.0I-PER1O
IN1.0O1O
SURPRISE1.0O1O
DEFEAT1.0O1O
1.0O1O
Nadim3.0I-PER1O
Ladki3.0I-PER1O
AL-AIN4.0I-LOC1O
1.0O1O

问题原因及解决方法

1. 数据预处理核心错误:序列构建完全不符合NER任务要求

NER是序列标注任务,需要以完整句子为单位构建序列,捕捉上下文依赖。但你当前把每个单词单独作为一个长度为100的序列(仅当前单词有效,其余为补0),LSTM根本无法学习到上下文信息,只能输出占比最高的O标签。

修正方式:按句子分组预处理数据

def load_and_preprocess_data(file_path):
    with open(file_path, 'r', encoding='utf-8') as file:
        conll_data = file.read().splitlines()
    
    sentences = []
    current_sentence = []
    for line in conll_data:
        line = line.strip()
        if not line:
            if current_sentence:
                sentences.append(current_sentence)
                current_sentence = []
        else:
            parts = line.split()
            word = parts[0].strip(string.punctuation)
            label = parts[-1]
            current_sentence.append((word, label))
    if current_sentence:
        sentences.append(current_sentence)
    
    return sentences

# 加载按句子分组的数据
train_sentences = load_and_preprocess_data('data/eng.train')
# 拆分单词序列和标签序列
train_texts = [[word for word, label in sent] for sent in train_sentences]
train_labels = [[label for word, label in sent] for sent in train_sentences]

2. 模型结构错误

  • 嵌入层后多余的Dense(8, activation='softmax')直接破坏了序列结构,必须删除。
  • LSTM默认返回序列最后一个时间步的结果,而NER需要每个时间步对应一个标签,需设置return_sequences=True,输出层要匹配序列长度的标签。

修正后的模型结构

model = Sequential()
model.add(Embedding(len(word_index) + 1, 
                    EMBEDDING_DIM, 
                    weights=[embedding_matrix], 
                    input_length=MAX_SEQUENCE_LENGTH, 
                    trainable=True))  # 改为可训练,适配NER任务
# 返回每个时间步的输出,用于序列标注
model.add(LSTM(LSTM_UNITS, dropout=0.2, recurrent_dropout=0.2, return_sequences=True))
# 每个时间步对应一个标签输出
model.add(Dense(len(labels_index), activation='softmax'))

model.compile(
    loss='categorical_crossentropy',
    optimizer='adam',
    metrics=['accuracy', F1Score(average='micro', sequence_len=MAX_SEQUENCE_LENGTH)]
)

3. 类别不平衡问题

CoNLL数据中O标签占比超过80%,模型会自然偏向预测该标签。可通过以下方式缓解:

  • 计算类别权重,在训练时赋予稀有标签更高权重:
from sklearn.utils.class_weight import compute_class_weight

# 扁平化标签列表计算权重
label_ids = [label_to_id[label] for sent_labels in train_labels for label in sent_labels]
class_weights = compute_class_weight('balanced', classes=np.unique(label_ids), y=label_ids)
class_weight_dict = {i: class_weights[i] for i in range(len(class_weights))}

# 训练时传入类别权重
history = model.fit(data, labels, epochs=EPOCHS, batch_size=BATCH_SIZE, 
                    validation_data=(x_val, y_val), class_weight=class_weight_dict)
  • 改用Focal Loss替代交叉熵损失,降低易分类样本的权重。

4. 嵌入层设置优化

原代码中嵌入层设置trainable=False,但通用预训练的Word2Vec不一定适配NER任务,改为trainable=True让嵌入层在任务上微调,能提升模型效果。


内容的提问来源于stack exchange,提问作者Chien Hui Lim

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.07 15:25:54