You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

视频人体姿态识别网络始终分类为同一类别问题求助

视频人体姿态识别神经网络训练问题排查求助

我正在构建一个视频人体姿态识别神经网络,数据集包含1000个姿态文件,每个文件对应一段视频,包含多帧关节关键点数据(比如torch.Size([256, 51, 45])代表[批量大小, 特征数, 帧数])。目前网络无法有效学习:损失仅在前几个epoch下降,之后趋于平稳,最终所有样本都被分类为类别2,推测是数据集类别2占比过高导致不平衡。

补充说明:数据集是npz文件,包含关节关键点的x坐标、y坐标及置信度,每个文件对应一段视频的多帧数据。

已尝试的解决方案

  • 构建两种模型:仅含少量1D CNN层的简单模型(Model A)和加入LSTM的模型(Model B),但结果几乎一致。
    Model A
    class ModelA(nn.Module):
        
        def __init__(self):
            super().__init__()
            self.hidden1 = nn.Conv1d(in_features, out_channels=64, kernel_size=3)
            self.hidden2 = nn.ReLU()
            self.hidden3 = nn.Conv1d(64, out_channels=128, kernel_size=3, stride = 2)
            self.hidden4 = nn.ReLU()
            self.hidden5 = nn.Conv1d(128, out_channels=256, kernel_size=3, stride = 2)
            self.hidden6 = nn.ReLU()
            self.hidden7 = nn.Conv1d(256, out_channels=512, kernel_size=3, stride = 2)
            self.hidden8 = nn.ReLU()
            self.hidden9 = nn.MaxPool1d(kernel_size=2)
            self.hidden10 = nn.Flatten()
            self.hidden11 = nn.Linear(1024, 100)
            #self.hidden12 = nn.Dropout(0.3)
            self.hidden13 = nn.ReLU()
            self.hidden14 = nn.Linear(100, n_outputs)
    
        def forward(self, x):
            x = self.hidden1(x)
            x = self.hidden2(x)
            x = self.hidden3(x)
            x = self.hidden4(x)
            x = self.hidden5(x)
            x = self.hidden6(x)
            x = self.hidden7(x)
            x = self.hidden8(x)
            x = self.hidden9(x)
            x = self.hidden10(x)
            x = self.hidden11(x)
            x = self.hidden12(x)
            x = self.hidden13(x)
            x = self.hidden13(x)
    
            return F.softmax(x, dim = 1)
    
    Model B
    class Dilated_blocks(nn.Module):
        def __init__(self, in_feat, out_feat, stride, dilation):
            super(Dilated_blocks, self).__init__()
            self.dilated_conv = nn.Conv1d(in_feat, out_feat, kernel_size = 3, stride = stride, dilation = dilation, padding = 1)
            self.conv_tranform = nn.Conv1d(out_feat, out_feat, kernel_size = 3, padding = 1)
    
        def forward(self, x):
            x = self.dilated_conv(x)
            x = self.conv_tranform(x)
    
            return x
    
    class ModelB(nn.Module):
        def __init__(self):
            super().__init__()
    
            self.conv1 = nn.Conv1d(in_channels = 51, out_channels = 64, kernel_size = 3) 
            self.batch1 = nn.BatchNorm1d(64)
    
            self.conv_block2 = Dilated_blocks(in_feat = 64,out_feat = 128, stride = 2, dilation = 2)
            self.batch2 = nn.BatchNorm1d(128)
    
            self.conv_block3 = Dilated_blocks(in_feat = 128, out_feat = 256, stride = 2, dilation = 2)
            self.batch3 = nn.BatchNorm1d(256)
    
            self.lstm_extractor = nn.LSTM(input_size=256, hidden_size=512, num_layers=10,
                                          dropout=0.2, batch_first=True)
    
    
            self.relu = nn.ReLU()
            
            self.pool = nn.MaxPool1d(2) #average, conv instead stride = 2, kernel = 2
            self.flat = nn.Flatten()
            self.fc1 = nn.Linear(512, 256)
            self.fc2 = nn.Linear(256, n_outputs)
    
    
    
    
        def forward(self, x):
            x = self.conv1(x)
            x = self.relu(x)
            x = self.batch1(x)
    
            #x = self.conv2(x) #your version
            x = self.conv_block2(x)
            x = self.relu(x)
            x = self.batch2(x)
    
            #x = self.conv3(x)
            x = self.conv_block3(x)
            x = self.relu(x)
            x = self.batch3(x)
    
            out = x.permute(0, 2, 1)
    
            out, (ht, ct) = self.lstm_extractor(out)
    
            out = ht[-1]
    
            out = self.flat(out)
    
            out = self.fc1(out)
            out = self.fc2(out)
            return F.softmax(out, dim = 1)
    
  • 针对数据集不平衡,尝试在交叉熵损失中加入类别权重(基于全数据集分布百分比计算),但性能无提升。损失与准确率曲线情况:
    • Model B无类别权重:准确率曲线前期上升后平稳,损失曲线前期下降后趋于水平
    • Model B含类别权重:准确率曲线无明显提升,损失曲线下降幅度依旧有限
  • 尝试不同学习率,无明显改善
  • 增加epoch数量后损失缓慢下降,但学习速度极慢,200epoch后损失仍较高

目前所有尝试均未解决问题,怀疑代码存在错误,附上训练循环及损失函数计算代码,恳请提供排查建议。

训练循环代码

for ep in range(1, num_epochs + 1):

            running_loss = 0.0
            accuracy = 0.0

            val_running_loss = 0.0
            val_accuracy = 0.0

            for iterat, data in enumerate(train_dataloader):

                #reset gradients of all model parameters to zero
                model.zero_grad()

                #set model in training mode
                model.train()

                #get input data: poses and labels 
                poses = data[0].type(torch.float).permute(0,2,1).to(device) 
                labels = data[1].type(torch.long).to(device) 


                #zero the parameter gradients
                optimizer.zero_grad()
                
                #getting labels in the right format for loss function
                most_frequent_values = torch.mode(labels, dim=1).values 
                #forward
                outputs = model(poses) 

                #loss 
                loss = criterion2(outputs, most_frequent_values)
                running_loss += loss.item()                
                
                #backward
                loss.backward()

                #optmize
                optimizer.step()

                #Accuracy
                acc = torchmetrics.functional.accuracy(outputs, most_frequent_values, task = 'multiclass', num_classes=8)
                accuracy += acc.item()
               


            #Validation
            with torch.no_grad():
                
                val_data = next(iter(val_dataloader))

                #set a model in evaluation mode. 
                model.eval()

                #get validation data
                val_poses = val_data[0].type(torch.float).permute(0,2,1).to(device)
                val_labels = val_data[1].type(torch.long).to(device)

                #labels in proper format
                val_most_frequent_values = torch.mode(val_labels, dim=1).values

                #forward
                val_outputs = model(val_poses)

                #loss
                val_loss = criterion2(val_outputs, val_most_frequent_values)
                val_running_loss += val_loss.item()

                #Accuracy
                val_acc = torchmetrics.functional.accuracy(val_outputs, val_most_frequent_values, task = 'multiclass', num_classes=8)
                val_accuracy += val_acc.item()



            #Training statistics for each epoch
            epoch_loss = running_loss / len (train_dataloader)
            train_loss.append(epoch_loss)
            epoch_acc = accuracy / len (train_dataloader)
            train_acc.append(epoch_acc)

            #Validation statistics for each epoch
            val_epoch_loss = val_running_loss 
            validation_loss.append(val_epoch_loss)
            val_epoch_acc = val_accuracy
            validation_acc.append(val_epoch_acc) 

损失函数及类别权重计算代码

classes_distribution = 1/get_classes_weights(whole_dataset)
criterion2 = nn.CrossEntropyLoss(weight = classes_distribution.to(device)) 
def get_classes_weights(dataloader):

    all_labels = []
        
    for iterati, data in enumerate(dataloader):
            
        labels = data[1].type(torch.long)
        most_frequent_values = torch.mode(labels, dim=1).values
        all_labels.append(most_frequent_values)        
            
    tensor_to_list = [item for tensor in all_labels for item in tensor.tolist()]
            
    number_of_classes = len(tensor_to_list)

    #Count the occurrences of each class
    classes_counts = {}
    for integer in tensor_to_list:
        classes_counts[integer] = classes_counts.get(integer, 0) + 1

    # Calculate the percentage distribution of each class
    percentage_distribution = {}
    for integer, count in classes_counts.items():
        percentage_distribution[integer] = (count / number_of_classes) 
               

    sorted_dict = {k: percentage_distribution[k] for k in sorted(percentage_distribution)}

    classes_weights = list(sorted_dict.values())

    return torch.tensor(classes_weights)

内容的提问来源于stack exchange,提问作者dlus

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.15 17:18:09