You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Django中cvzone HandDetector结合摄像头实现手势识别遇阻求助

Django摄像头石头剪刀布项目:手部检测与视频处理问题

我正在开发一个基于Django的石头剪刀布游戏,通过摄像头和电脑对战。目前用HandDetector检测摄像头流中的手部,Keras分类器做手势分类,但遇到了HandDetector在摄像头视频中无法正常工作的问题。

试过发送单帧图片到后端,但HandDetector识别效果很差。打算改用视频流处理,但不确定在Django里怎么实现。


前端捕获与上传代码

<div class="container">
  <div class="row">
    <div class="col-md-6">
      <video id="webcam" width="640" height="480" autoplay></video>
      <div id="countdown" style="font-size: 48px; text-align: center;"></div>
      <div id="result" style="margin-top: 20px; font-size: 24px;"></div>
    </div>
    <div class="col-md-6">
      <button id="play-button" class="btn btn-primary" onclick="play()">Play</button>
    </div>
  </div>
</div>

<script src="https://code.jquery.com/jquery-3.6.0.min.js"></script>

<script>
$(document).ready(function() { startWebcam(); });

var video = document.getElementById('webcam');
var stream;

function startWebcam() {
    navigator.mediaDevices.getUserMedia({ video: true, audio: false })
    .then(function(localMediaStream) {
        stream = localMediaStream;
        video.srcObject = localMediaStream;
        video.play();
    })
    .catch(function(err) {
        console.log("An error occurred: " + err);
    });
}

function play() {
    // 开始录制
    var chunks = [];
    var recorder = new MediaRecorder(stream);
    recorder.start();

    // 3秒倒计时
    var count = 3;
    var countdown = setInterval(function() {
        $('#countdown').html(count);
        count--;
        if (count === -1) {
            clearInterval(countdown);
            recorder.stop();
            // 录制结束后停止摄像头
            var tracks = stream.getTracks();
            tracks.forEach(track => track.stop());
        }
    }, 1000);

    recorder.ondataavailable = function(e) {
        chunks.push(e.data);
    }

    recorder.onstop = function() {
        var blob = new Blob(chunks, { type: 'video/webm' });
        chunks = [];

        var formData = new FormData();
        formData.append("video_data", blob);
        formData.append("csrfmiddlewaretoken", '{{ csrf_token }}');

        $.ajax({
            type: 'POST',
            url: '/process-video/',
            data: formData,
            processData: false,
            contentType: false,
            success: function(response) {
                console.log(response);
                $('#result').html("你的手势:" + response);
                // 重新启动摄像头,方便下一次游戏
                startWebcam();
            },
            error: function(xhr, status, error) {
                console.log(error);
                $('#result').html("检测失败,请重试");
                startWebcam();
            }
        });
    };
}
</script>

Django后端视图代码

import cv2
import math
from collections import Counter
from django.http import HttpResponse
from cvzone.HandTrackingModule import HandDetector
from .hand_recognition.HandClassifier import HandClassifier
from .hand_recognition.ImageProcessor import ImageProcessor

# 全局初始化实例,避免重复创建浪费资源
detector = HandDetector(maxHands=1)
classifier = HandClassifier()
processor = ImageProcessor(detector, 300, 20)

def process_video(request):
    if request.method != 'POST':
        return HttpResponse("Error: Invalid request method")

    video_data = request.FILES.get('video_data')
    if not video_data:
        return HttpResponse("Error: No video data received")

    cap = cv2.VideoCapture(video_data)
    frame_count = 0
    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
    predictions = []

    # 取视频中间1/3的帧处理,避开开头结尾的无效帧
    start_frame = total_frames // 3
    end_frame = total_frames * 2 // 3

    while cap.isOpened():
        ret, frame = cap.read()
        if not ret:
            break
        frame_count +=1
        if frame_count < start_frame or frame_count > end_frame:
            continue
        
        prediction = process_image_data(frame)
        if prediction != "Error: No hand detected in the image":
            predictions.append(prediction)
    
    cap.release()

    if predictions:
        # 取出现次数最多的预测结果
        most_common = Counter(predictions).most_common(1)[0][0]
        return HttpResponse(most_common)
    else:
        return HttpResponse("Error: No hand detected in the video. Please try again.")


def process_image_data(image_data):
    global detector, processor, classifier
    hands = detector.findHands(image_data, draw=False)

    if hands:
        hand = hands[0]
        x, y, w, h = hand['bbox']
        img_white, img_resize = processor.process(image_data, hand)
        aspect_ratio = h / w

        if aspect_ratio > 1:
            h_gap = math.ceil((300 - img_resize.shape[0]) / 2)
            img_white[h_gap:h_gap + img_resize.shape[0], :] = img_resize
        else:
            w_gap = math.ceil((300 - img_resize.shape[1]) / 2)
            img_white[:, w_gap:w_gap + img_resize.shape[1]] = img_resize

        return classifier.classify(img_white)
    else:
        return "Error: No hand detected in the image"

HandClassifier类代码

import cv2
import numpy as np
from PIL import Image, ImageOps
from keras.models import load_model
import os


class HandClassifier:
    def __init__(self):
        # 修正路径,确保项目部署时能找到模型文件
        base_dir = os.path.dirname(os.path.abspath(__file__))
        model_path = os.path.join(base_dir, '../keras/keras_model.h5')
        labels_path = os.path.join(base_dir, '../keras/labels.txt')

        self.model = load_model(model_path, compile=False)
        with open(labels_path, 'r') as f:
            self.labels = f.read().splitlines()

    def classify(self, image):
        # 修复convert未赋值的问题,确保转为RGB格式
        image = Image.fromarray(image).convert('RGB')
        size = (224, 224)
        image = ImageOps.fit(image, size, Image.Resampling.LANCZOS)
        
        image_array = np.asarray(image)
        normalized_image_array = (image_array.astype(np.float32) / 127.0) - 1
        data = np.ndarray(shape=(1, 224, 224, 3), dtype=np.float32)
        data[0] = normalized_image_array
        
        # 关闭verbose减少冗余输出,去掉重复的predict调用
        prediction = self.model.predict(data, verbose=0)
        index = np.argmax(prediction)
        class_name = self.labels[index]

        return class_name

ImageProcessor类代码

import cv2
import numpy as np


class ImageProcessor:
    def __init__(self, detector, img_size, offset):
        self.detector = detector
        self.img_size = img_size
        self.offset = offset

    def process(self, img, hand):
        x, y, w, h = hand['bbox']
        # 防止裁剪超出图像边界
        y_start = max(0, y - self.offset)
        y_end = min(img.shape[0], y + h + self.offset)
        x_start = max(0, x - self.offset)
        x_end = min(img.shape[1], x + w + self.offset)
        
        img_crop = img[y_start:y_end, x_start:x_end]

        img_crop = cv2.cvtColor(img_crop, cv2.COLOR_BGR2GRAY)
        img_hog = cv2.normalize(img_crop, None, alpha=0, beta=255, norm_type=cv2.NORM_MINMAX, dtype=cv2.CV_8U)
        img_hog = cv2.medianBlur(img_hog, ksize=5)

        img_crop = cv2.resize(img_hog, (self.img_size, self.img_size))
        img_resize = np.expand_dims(img_crop, axis=2)
        img_white = np.ones((self.img_size, self.img_size, 3), np.uint8) * 255

        return img_white, img_resize

本地测试代码

import cv2
from cvzone.HandTrackingModule import HandDetector
import math

from src.rps.hand_recognition.HandClassifier import HandClassifier
from src.rps.hand_recognition.ImageProcessor import ImageProcessor

cap = cv2.VideoCapture(0)
detector = HandDetector(maxHands=1)
classifier = HandClassifier()

offset = 20
img_size = 300

labels = ["paper", "rock", "scissors"]

while True:
    processor = ImageProcessor(detector, img_size, offset)
    success, img = cap.read()
    hands = detector.findHands(img, draw=False)

    if hands:
        hand = hands[0]
        x, y, w, h = hand['bbox']
        img_white, img_resize = processor.process(img, hand)
        aspect_ratio = h / w

        if aspect_ratio > 1:
            h_gap = math.ceil((img_size - img_resize.shape[0]) / 2)
            img_white[h_gap:h_gap + img_resize.shape[0], :] = img_resize
        else:
            w_gap = math.ceil((img_size - img_resize.shape[1]) / 2)
            img_white[:, w_gap:w_gap + img_resize.shape[1]] = img_resize

        prediction = classifier.classify(img_white)

        cv2.rectangle(img, (x - offset, y - offset - 50),
                      (x - offset + 90, y - offset - 50 + 50), (255, 0, 255), cv2.FILLED)
        cv2.putText(img, prediction, (x, y - 26), cv2.FONT_HERSHEY_COMPLEX, 1.7, (255, 255, 255), 2)
        cv2.rectangle(img, (x - offset, y - offset),
                      (x + w + offset, y + h + offset), (255, 0, 255), 4)

        cv2.imshow("ImageWhite", img_white)

    cv2.imshow("Image", img)
    cv2.waitKey(1)

核心优化点说明

  1. 前端录制逻辑修复:原代码提前停止摄像头track导致录制空视频,改为录制结束后再停止摄像头,同时增加游戏后重启摄像头的逻辑
  2. 后端资源复用:全局初始化检测器和分类器,避免每次请求重复创建实例,大幅提升性能
  3. 视频帧筛选:只处理视频中间1/3的帧,避开开头、结尾的无效帧,提升检测准确率
  4. 图像边界保护:修复ImageProcessor中裁剪超出图像边界的问题
  5. 分类器效率优化:去掉重复的predict调用,关闭verbose输出,修复RGB格式转换的赋值问题

内容的提问来源于stack exchange,提问作者Seppe Willems

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.05 15:15:29