You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用OpenCV Python的VideoCapture从不同帧数视频取30帧且不丢数据

解决方案:均匀采样+补帧处理

要在不遗漏动作信息的前提下为每个视频获取固定30帧,最合理的方式是对视频帧进行均匀采样,让选取的帧覆盖整个手语动作的全过程;如果视频本身帧数不足30,则通过补帧(重复末尾帧)来凑够数量。

具体实现思路

  1. 先获取视频总帧数,分两种情况处理:
    • 当视频帧数≥30时:计算采样步长,均匀选取30帧,确保覆盖动作的开始、中间和结束阶段
    • 当视频帧数<30时:先提取所有有效帧,再重复最后几帧直到凑够30帧,避免丢失动作信息
  2. 直接跳转到目标帧读取,减少逐帧遍历的冗余计算,提升处理效率

修改后的完整代码

import cv2
import numpy as np
import os
import mediapipe as mp

DATASET_PATH = "/home/kuna71/Dev/HearMySign/Datasets/Adjectives_1of8/Adjectives"
KEYPOINT_PATH = "/home/kuna71/Dev/HearMySign/Keypoints"
sequence_len = 30

mp_holistic = mp.solutions.holistic
mp_drawing = mp.solutions.drawing_utils

def mediapipe_detection(image, model):
    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)
    image.flags.writeable = False
    results = model.process(image)
    image.flags.writeable = True
    image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)
    return image, results

def extract_keypoints(results):
    pose = np.array([[res.x, res.y, res.z, res.visibility] for res in results.pose_landmarks.landmark]).flatten() if results.pose_landmarks else np.zeros(33*4)
    face = np.array([[res.x, res.y, res.z] for res in results.face_landmarks.landmark]).flatten() if results.face_landmarks else np.zeros(468*3)
    lh = np.array([[res.x, res.y, res.z] for res in results.left_hand_landmarks.landmark]).flatten() if results.left_hand_landmarks else np.zeros(21*3)
    rh = np.array([[res.x, res.y, res.z] for res in results.right_hand_landmarks.landmark]).flatten() if results.right_hand_landmarks else np.zeros(21*3)
    return np.concatenate([pose, face, lh, rh])

# 遍历数据集目录
directories = os.listdir(DATASET_PATH)
for d in directories:
    vids = os.listdir(os.path.join(DATASET_PATH, d))
    for v in vids:
        videopath = os.path.join(DATASET_PATH, d, v)
        print(f"\n\n处理视频:{videopath}")
        
        cap = cv2.VideoCapture(videopath)
        length = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
        print(f"视频总帧数:{length}")
        
        # 创建关键点保存目录
        save_dir = os.path.join(KEYPOINT_PATH, d, v)
        os.makedirs(save_dir, exist_ok=True)
        
        # 初始化Mediapipe模型
        with mp_holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.5) as holistic:
            keypoints_sequence = []
            
            if length >= sequence_len:
                # 均匀采样30帧:计算采样步长,确保覆盖全视频
                step = length // sequence_len
                # 生成采样帧索引,最后一帧固定为视频末尾
                frame_indices = [i * step for i in range(sequence_len)]
                frame_indices[-1] = length - 1
                
                for idx in frame_indices:
                    # 跳转到目标帧读取
                    cap.set(cv2.CAP_PROP_POS_FRAMES, idx)
                    ret, frame = cap.read()
                    if not ret:
                        continue
                    image, results = mediapipe_detection(frame, holistic)
                    keypoints = extract_keypoints(results)
                    keypoints_sequence.append(keypoints)
                    
                    # 可选:显示当前处理帧
                    cv2.imshow('OpenCV Feed', image)
                    if cv2.waitKey(10) & 0xFF == ord('q'):
                        break
            else:
                # 帧数不足30,先读取所有有效帧
                while cap.isOpened():
                    ret, frame = cap.read()
                    if not ret:
                        break
                    image, results = mediapipe_detection(frame, holistic)
                    keypoints = extract_keypoints(results)
                    keypoints_sequence.append(keypoints)
                    
                    cv2.imshow('OpenCV Feed', image)
                    if cv2.waitKey(10) & 0xFF == ord('q'):
                        break
                # 重复最后一帧直到凑够30帧
                while len(keypoints_sequence) < sequence_len:
                    keypoints_sequence.append(keypoints_sequence[-1])
            
            # 保存每帧关键点到对应目录
            for i, kp in enumerate(keypoints_sequence, 1):
                npy_path = os.path.join(save_dir, str(i))
                np.save(npy_path, kp)
        
        cap.release()
cv2.destroyAllWindows()

关键优化说明

  • 均匀采样:通过步长计算让选取的帧均匀分布在整个视频中,避免只截取开头动作,完整保留手语动作的时序特征
  • 补帧逻辑:针对短视频重复末尾帧,既不丢失现有动作信息,又满足LSTM模型对输入序列长度的固定要求
  • 效率提升:直接跳转到目标帧读取,减少了逐帧遍历的冗余计算,加快处理速度

内容的提问来源于stack exchange,提问作者Kunal Kankaria

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.19 11:35:21