视频人体姿态识别网络始终分类为同一类别问题求助
视频人体姿态识别神经网络训练问题排查求助
我正在构建一个视频人体姿态识别神经网络,数据集包含1000个姿态文件,每个文件对应一段视频,包含多帧关节关键点数据(比如torch.Size([256, 51, 45])代表[批量大小, 特征数, 帧数])。目前网络无法有效学习:损失仅在前几个epoch下降,之后趋于平稳,最终所有样本都被分类为类别2,推测是数据集类别2占比过高导致不平衡。
补充说明:数据集是npz文件,包含关节关键点的x坐标、y坐标及置信度,每个文件对应一段视频的多帧数据。
已尝试的解决方案
- 构建两种模型:仅含少量1D CNN层的简单模型(Model A)和加入LSTM的模型(Model B),但结果几乎一致。
Model A
Model Bclass ModelA(nn.Module): def __init__(self): super().__init__() self.hidden1 = nn.Conv1d(in_features, out_channels=64, kernel_size=3) self.hidden2 = nn.ReLU() self.hidden3 = nn.Conv1d(64, out_channels=128, kernel_size=3, stride = 2) self.hidden4 = nn.ReLU() self.hidden5 = nn.Conv1d(128, out_channels=256, kernel_size=3, stride = 2) self.hidden6 = nn.ReLU() self.hidden7 = nn.Conv1d(256, out_channels=512, kernel_size=3, stride = 2) self.hidden8 = nn.ReLU() self.hidden9 = nn.MaxPool1d(kernel_size=2) self.hidden10 = nn.Flatten() self.hidden11 = nn.Linear(1024, 100) #self.hidden12 = nn.Dropout(0.3) self.hidden13 = nn.ReLU() self.hidden14 = nn.Linear(100, n_outputs) def forward(self, x): x = self.hidden1(x) x = self.hidden2(x) x = self.hidden3(x) x = self.hidden4(x) x = self.hidden5(x) x = self.hidden6(x) x = self.hidden7(x) x = self.hidden8(x) x = self.hidden9(x) x = self.hidden10(x) x = self.hidden11(x) x = self.hidden12(x) x = self.hidden13(x) x = self.hidden13(x) return F.softmax(x, dim = 1)class Dilated_blocks(nn.Module): def __init__(self, in_feat, out_feat, stride, dilation): super(Dilated_blocks, self).__init__() self.dilated_conv = nn.Conv1d(in_feat, out_feat, kernel_size = 3, stride = stride, dilation = dilation, padding = 1) self.conv_tranform = nn.Conv1d(out_feat, out_feat, kernel_size = 3, padding = 1) def forward(self, x): x = self.dilated_conv(x) x = self.conv_tranform(x) return x class ModelB(nn.Module): def __init__(self): super().__init__() self.conv1 = nn.Conv1d(in_channels = 51, out_channels = 64, kernel_size = 3) self.batch1 = nn.BatchNorm1d(64) self.conv_block2 = Dilated_blocks(in_feat = 64,out_feat = 128, stride = 2, dilation = 2) self.batch2 = nn.BatchNorm1d(128) self.conv_block3 = Dilated_blocks(in_feat = 128, out_feat = 256, stride = 2, dilation = 2) self.batch3 = nn.BatchNorm1d(256) self.lstm_extractor = nn.LSTM(input_size=256, hidden_size=512, num_layers=10, dropout=0.2, batch_first=True) self.relu = nn.ReLU() self.pool = nn.MaxPool1d(2) #average, conv instead stride = 2, kernel = 2 self.flat = nn.Flatten() self.fc1 = nn.Linear(512, 256) self.fc2 = nn.Linear(256, n_outputs) def forward(self, x): x = self.conv1(x) x = self.relu(x) x = self.batch1(x) #x = self.conv2(x) #your version x = self.conv_block2(x) x = self.relu(x) x = self.batch2(x) #x = self.conv3(x) x = self.conv_block3(x) x = self.relu(x) x = self.batch3(x) out = x.permute(0, 2, 1) out, (ht, ct) = self.lstm_extractor(out) out = ht[-1] out = self.flat(out) out = self.fc1(out) out = self.fc2(out) return F.softmax(out, dim = 1) - 针对数据集不平衡,尝试在交叉熵损失中加入类别权重(基于全数据集分布百分比计算),但性能无提升。损失与准确率曲线情况:
- Model B无类别权重:准确率曲线前期上升后平稳,损失曲线前期下降后趋于水平
- Model B含类别权重:准确率曲线无明显提升,损失曲线下降幅度依旧有限
- 尝试不同学习率,无明显改善
- 增加epoch数量后损失缓慢下降,但学习速度极慢,200epoch后损失仍较高
目前所有尝试均未解决问题,怀疑代码存在错误,附上训练循环及损失函数计算代码,恳请提供排查建议。
训练循环代码
for ep in range(1, num_epochs + 1): running_loss = 0.0 accuracy = 0.0 val_running_loss = 0.0 val_accuracy = 0.0 for iterat, data in enumerate(train_dataloader): #reset gradients of all model parameters to zero model.zero_grad() #set model in training mode model.train() #get input data: poses and labels poses = data[0].type(torch.float).permute(0,2,1).to(device) labels = data[1].type(torch.long).to(device) #zero the parameter gradients optimizer.zero_grad() #getting labels in the right format for loss function most_frequent_values = torch.mode(labels, dim=1).values #forward outputs = model(poses) #loss loss = criterion2(outputs, most_frequent_values) running_loss += loss.item() #backward loss.backward() #optmize optimizer.step() #Accuracy acc = torchmetrics.functional.accuracy(outputs, most_frequent_values, task = 'multiclass', num_classes=8) accuracy += acc.item() #Validation with torch.no_grad(): val_data = next(iter(val_dataloader)) #set a model in evaluation mode. model.eval() #get validation data val_poses = val_data[0].type(torch.float).permute(0,2,1).to(device) val_labels = val_data[1].type(torch.long).to(device) #labels in proper format val_most_frequent_values = torch.mode(val_labels, dim=1).values #forward val_outputs = model(val_poses) #loss val_loss = criterion2(val_outputs, val_most_frequent_values) val_running_loss += val_loss.item() #Accuracy val_acc = torchmetrics.functional.accuracy(val_outputs, val_most_frequent_values, task = 'multiclass', num_classes=8) val_accuracy += val_acc.item() #Training statistics for each epoch epoch_loss = running_loss / len (train_dataloader) train_loss.append(epoch_loss) epoch_acc = accuracy / len (train_dataloader) train_acc.append(epoch_acc) #Validation statistics for each epoch val_epoch_loss = val_running_loss validation_loss.append(val_epoch_loss) val_epoch_acc = val_accuracy validation_acc.append(val_epoch_acc)
损失函数及类别权重计算代码
classes_distribution = 1/get_classes_weights(whole_dataset) criterion2 = nn.CrossEntropyLoss(weight = classes_distribution.to(device))
def get_classes_weights(dataloader): all_labels = [] for iterati, data in enumerate(dataloader): labels = data[1].type(torch.long) most_frequent_values = torch.mode(labels, dim=1).values all_labels.append(most_frequent_values) tensor_to_list = [item for tensor in all_labels for item in tensor.tolist()] number_of_classes = len(tensor_to_list) #Count the occurrences of each class classes_counts = {} for integer in tensor_to_list: classes_counts[integer] = classes_counts.get(integer, 0) + 1 # Calculate the percentage distribution of each class percentage_distribution = {} for integer, count in classes_counts.items(): percentage_distribution[integer] = (count / number_of_classes) sorted_dict = {k: percentage_distribution[k] for k in sorted(percentage_distribution)} classes_weights = list(sorted_dict.values()) return torch.tensor(classes_weights)
内容的提问来源于stack exchange,提问作者dlus
相关产品推荐
相关产品推荐

