You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何将实时摄像头MediaPipe动作检测代码集成到Django项目

Django集成MediaPipe实时动作检测视频流实现方案

1. camera.py 实现

将动作检测逻辑封装为类,核心是生成可被Web响应解析的视频帧(替代原代码中cv2.imshow的本地窗口显示方式):

import cv2
import numpy as np
import mediapipe as mp

# 实现/导入动作检测所需辅助函数
def mediapipe_detection(image, model):
    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)
    image.flags.writeable = False
    results = model.process(image)
    image.flags.writeable = True
    image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)
    return image, results

def draw_styled_landmarks(image, results):
    # 绘制面部关键点
    mp.solutions.drawing_utils.draw_landmarks(
        image, results.face_landmarks, mp.solutions.holistic.FACEMESH_TESSELATION,
        mp.solutions.drawing_utils.DrawingSpec(color=(80,110,10), thickness=1, circle_radius=1),
        mp.solutions.drawing_utils.DrawingSpec(color=(80,256,121), thickness=1, circle_radius=1)
    )
    # 绘制姿态关键点
    mp.solutions.drawing_utils.draw_landmarks(
        image, results.pose_landmarks, mp.solutions.holistic.POSE_CONNECTIONS,
        mp.solutions.drawing_utils.DrawingSpec(color=(80,22,10), thickness=2, circle_radius=4),
        mp.solutions.drawing_utils.DrawingSpec(color=(80,44,121), thickness=2, circle_radius=2)
    )
    # 绘制左右手关键点
    mp.solutions.drawing_utils.draw_landmarks(
        image, results.left_hand_landmarks, mp.solutions.holistic.HAND_CONNECTIONS,
        mp.solutions.drawing_utils.DrawingSpec(color=(121,22,76), thickness=2, circle_radius=4),
        mp.solutions.drawing_utils.DrawingSpec(color=(121,44,250), thickness=2, circle_radius=2)
    )
    mp.solutions.drawing_utils.draw_landmarks(
        image, results.right_hand_landmarks, mp.solutions.holistic.HAND_CONNECTIONS,
        mp.solutions.drawing_utils.DrawingSpec(color=(245,117,66), thickness=2, circle_radius=4),
        mp.solutions.drawing_utils.DrawingSpec(color=(245,66,230), thickness=2, circle_radius=2)
    )

def extract_keypoints(results):
    pose = np.array([[res.x, res.y, res.z, res.visibility] for res in results.pose_landmarks.landmark]).flatten() if results.pose_landmarks else np.zeros(33*4)
    face = np.array([[res.x, res.y, res.z] for res in results.face_landmarks.landmark]).flatten() if results.face_landmarks else np.zeros(468*3)
    lh = np.array([[res.x, res.y, res.z] for res in results.left_hand_landmarks.landmark]).flatten() if results.left_hand_landmarks else np.zeros(21*3)
    rh = np.array([[res.x, res.y, res.z] for res in results.right_hand_landmarks.landmark]).flatten() if results.right_hand_landmarks else np.zeros(21*3)
    return np.concatenate([pose, face, lh, rh])

def prob_viz(res, actions, input_frame, colors):
    output_frame = input_frame.copy()
    for num, prob in enumerate(res):
        cv2.rectangle(output_frame, (0,60+num*40), (int(prob*100), 90+num*40), colors[num], -1)
        cv2.putText(output_frame, actions[num], (0, 85+num*40), cv2.FONT_HERSHEY_SIMPLEX, 1, (255,255,255), 2, cv2.LINE_AA)
    return output_frame

class ActionDetectionCamera:
    def __init__(self):
        # 初始化检测状态变量
        self.sequence = []
        self.sentence = []
        self.predictions = []
        self.threshold = 0.5
        # 替换为你的动作列表、模型路径、颜色配置
        self.actions = np.array(['hello', 'thanks', 'iloveyou'])
        self.colors = [(245,117,16), (117,245,16), (16,117,245)]
        # 加载预训练模型
        from tensorflow.keras.models import load_model
        self.model = load_model('action.h5')
        # 初始化MediaPipe模型与摄像头
        self.holistic = mp.solutions.holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.5)
        self.cap = cv2.VideoCapture(0)

    def __del__(self):
        self.cap.release()
        cv2.destroyAllWindows()

    def get_frame(self):
        ret, frame = self.cap.read()
        if not ret:
            return None
        
        # 执行动作检测流程
        image, results = mediapipe_detection(frame, self.holistic)
        draw_styled_landmarks(image, results)

        keypoints = extract_keypoints(results)
        self.sequence.append(keypoints)
        self.sequence = self.sequence[-30:]

        if len(self.sequence) == 30:
            res = self.model.predict(np.expand_dims(self.sequence, axis=0))[0]
            self.predictions.append(np.argmax(res))

            # 结果判定与可视化
            if np.unique(self.predictions[-10:])[0] == np.argmax(res):
                if res[np.argmax(res)] > self.threshold:
                    if len(self.sentence) > 0:
                        if self.actions[np.argmax(res)] != self.sentence[-1]:
                            self.sentence.append(self.actions[np.argmax(res)])
                    else:
                        self.sentence.append(self.actions[np.argmax(res)])

            if len(self.sentence) > 5:
                self.sentence = self.sentence[-5:]

            image = prob_viz(res, self.actions, image, self.colors)

        # 绘制结果文本栏
        cv2.rectangle(image, (0,0), (640, 40), (245, 117, 16), -1)
        cv2.putText(image, ' '.join(self.sentence), (3,30), 
                    cv2.FONT_HERSHEY_SIMPLEX, 1, (255, 255, 255), 2, cv2.LINE_AA)

        # 将帧编码为JPEG格式返回
        ret, jpeg = cv2.imencode('.jpg', image)
        return jpeg.tobytes()

2. views.py 实现

编写视频流生成视图与主页视图:

from django.http import StreamingHttpResponse
from django.shortcuts import render
from .camera import ActionDetectionCamera
import time

def gen(camera):
    while True:
        frame = camera.get_frame()
        if frame is None:
            time.sleep(0.1)
            continue
        # 按HTTP流式传输格式返回帧数据
        yield (b'--frame\r\n'
               b'Content-Type: image/jpeg\r\n\r\n' + frame + b'\r\n\r\n')

def video_feed(request):
    return StreamingHttpResponse(gen(ActionDetectionCamera()),
                                 content_type='multipart/x-mixed-replace; boundary=frame')

def index(request):
    return render(request, 'action_detection/index.html')

3. urls.py 实现

配置URL路由关联视图:

from django.urls import path
from . import views

urlpatterns = [
    path('', views.index, name='index'),
    path('video_feed/', views.video_feed, name='video_feed'),
]

4. HTML模板(templates/action_detection/index.html)

创建显示视频流的前端页面:

<!DOCTYPE html>
<html>
<head>
    <title>实时动作检测</title>
    <style>
        body {
            display: flex;
            justify-content: center;
            align-items: center;
            min-height: 100vh;
            margin: 0;
            background-color: #f0f0f0;
        }
        .video-container {
            border: 2px solid #245117;
            border-radius: 8px;
            overflow: hidden;
        }
    </style>
</head>
<body>
    <div class="video-container">
        <img src="{% url 'video_feed' %}" alt="实时动作检测视频流">
    </div>
</body>
</html>

关键注意事项

  • 确保安装依赖库:pip install mediapipe opencv-python tensorflow django
  • 模型路径需与项目实际路径匹配,建议使用项目根目录相对路径
  • 本地测试需确保摄像头权限正常;服务器部署需确认服务器可访问硬件摄像头
  • 可优化模型加载逻辑,避免每个请求重复加载模型(如将模型加载到全局变量)

内容的提问来源于stack exchange,提问作者samuelkaris

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.28 11:05:24