You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于深度学习模型:摄像头帧替代JPG图像做手语预测遇异常

手语CNN模型实时预测异常排查

我用Keras搭建了手语识别CNN模型,结合OpenCV做实时翻译时遇到问题:测试文件夹里的JPG图片预测完全准确,但摄像头捕获的帧始终被预测为[27](对应标签nothing),明明帧的处理流程和图片一致,找不到问题所在。


测试图片的可正常运行代码

imgs_dir = r'C:\Users\danie\sign-language-alpha\data\asl_alphabet_test\asl_alphabet_test'
imgs = os.listdir(imgs_dir)

class_mapping = train_images.class_indices

def get_class_label(predictions, class_mapping):
    labels_mapping = {v: k for k, v in class_mapping.items()}
    predicted_labels = [labels_mapping[pred] for pred in predictions]

    return predicted_labels

def predict_image_class(model, img_path, class_mapping):
    img = cv2.imread(os.path.join(imgs_dir, img_path))
    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
    img = np.expand_dims(cv2.resize(img, (64, 64)), axis = 0)
    predictions = np.argmax(model.predict(img, verbose = 0), axis = 1)
    predicted_labels = get_class_label(predictions, class_mapping)
    return predicted_labels

import cv2
import numpy as np

def process_and_save_video_frame(frame, save_path="processed_frame.jpg"):
    """
    Process a video frame: convert to grayscale, resize with aspect ratio,
    normalize, reshape, and save the processed frame.
    
    Args:
        frame: The original video frame captured from the webcam.
        save_path: The path to save the processed frame image.
    
    Returns:
        frame_final: The processed frame ready for model input.
    """
    def resize_with_aspect_ratio(image, target_size):
        h, w = image.shape[:2]
        target_w, target_h = target_size
        scale = min(target_w / w, target_h / h)
        new_w, new_h = int(w * scale), int(h * scale)
        resized = cv2.resize(image, (new_w, new_h))
        delta_w, delta_h = target_w - new_w, target_h - new_h
        top, bottom = delta_h // 2, delta_h - (delta_h // 2)
        left, right = delta_w // 2, delta_w - (delta_w // 2)
        color = [0, 0, 0]
        new_image = cv2.copyMakeBorder(resized, top, bottom, left, right, cv2.BORDER_CONSTANT, value=color)
        return new_image

    print(f"Original Frame shape: {frame.shape}")
    
    # Convert to grayscale
    frame_gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
    
    # Resize with aspect ratio
    frame_resized = resize_with_aspect_ratio(frame_gray, (64, 64))
    print(f"Shape after resizing with aspect ratio to 64x64: {frame_resized.shape}")
    
    # Normalize
    frame_normalized = frame_resized.astype('float32') / 255.0
    
    # Add batch and channel dimensions
    frame_final = np.expand_dims(frame_normalized, axis=0)
    frame_final = np.expand_dims(frame_final, axis=-1)
    print(f"Shape after adding batch and channel dimensions: {frame_final.shape}")
    
    # Save the processed frame
    cv2.imwrite(save_path, frame_resized * 255)
    
    return frame_final


for img in imgs:
    predicted_labels = predict_image_class(model, img, class_mapping)
    print(f'Predicted Labels {img} -----> {predicted_labels}')

摄像头帧处理的异常代码

import os
import cv2
import numpy as np
from keras.models import load_model

model = load_model("FINAL.h5")
imgs_dir = r'C:\Users\danie\sign-language-alpha\data\asl_alphabet_test\asl_alphabet_test'
imgs = os.listdir(imgs_dir)

class_mapping = train_images.class_indices
print(f"Class mapping: {class_mapping}")

def get_class_label(predictions, class_mapping):
    labels_mapping = {v: k for k, v in class_mapping.items()}
    predicted_labels = [labels_mapping[pred] for pred in predictions]
    print(predicted_labels)
    return predicted_labels

def predict_image_class_from_path(model, img_path, class_mapping):
    img = cv2.imread(img_path)
    if img is None:
        print(f"Error: Unable to load image at path {img_path}")
        return ["Error"]
    print(f"Original image shape: {img.shape}")
    # Convert to grayscale
    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
    print(f"Shape after converting to grayscale: {img.shape}")
    # Resize the image to 64x64
    img = cv2.resize(img, (64, 64))
    print(f"Shape after resizing to 64x64: {img.shape}")
    # Normalize the image
    img = img.astype('float32') / 255.0
    # Add batch dimension and channel dimension
    img = np.expand_dims(img, axis=0)
    img = np.expand_dims(img, axis=-1)
    print(f"Shape after adding batch and channel dimensions: {img.shape}")
    # Make predictions
    predictions = np.argmax(model.predict(img, verbose=0), axis=1)
    print(f"Predictions: {predictions}")
    # Get the predicted label
    predicted_labels = get_class_label(predictions, class_mapping)
    return predicted_labels

def predict_image_class_from_frame(model, frame, class_mapping):
    # Convert to grayscale
    img = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
    print(f"Shape after converting to grayscale: {img.shape}")
    # Resize the image to 64x64
    img = cv2.resize(img, (64, 64))
    print(f"Shape after resizing to 64x64: {img.shape}")
    # Normalize the image
    img = img.astype('float32') / 255.0
    # Add batch dimension and channel dimension
    img = np.expand_dims(img, axis=0)
    img = np.expand_dims(img, axis=-1)
    print(f"Shape after adding batch and channel dimensions: {img.shape}")
    # Make predictions
    predictions = np.argmax(model.predict(img, verbose=0), axis=1)
    print(f"Predictions: {predictions}")
    # Get the predicted label
    predicted_labels = get_class_label(predictions, class_mapping)
    return predicted_labels
cap = cv2.VideoCapture(0)

while True:
    ret, frame = cap.read()
    if not ret:
        break

    frame_resized = cv2.resize(frame, (200, 200))
    predicted_labels = predict_image_class_from_frame(model, frame_resized, class_mapping)
    
    # Display the predicted label on the frame
    cv2.putText(frame_resized, predicted_labels[0], (50, 50), cv2.FONT_HERSHEY_SIMPLEX, 1, (0, 255, 0), 2, cv2.LINE_AA)
    
    # Display the frame
    cv2.imshow('Sign Language Prediction', frame_resized)
    
    # Break the loop on 'q' key press
    if cv2.waitKey(1) & 0xFF == ord('q'):
        break

 #{'A': 0, 'B': 1, 'C': 2, 'D': 3, 'E': 4, 'F': 5, 'G': 6, 'H': 7, 'I': 8, 'J': 9, 'K': 10, 'L': 11, 'M': 12, 'N': 13, 'O': 14, 'P': 15, 'Q': 16, 'R': 17, 'S': 18, 'T': 19, 'U': 20, 'V': 21, 'W': 22, 'X': 23, 'Y': 24, 'Z': 25, 'del': 26, 'nothing': 27, 'space': 28}

cap.release()
cv2.destroyAllWindows()

问题排查与解决

1. 预处理流程不一致

测试代码和摄像头帧处理的核心差异:

  • 测试代码未做归一化(无/255.0),但摄像头处理做了归一化,导致输入数据分布和模型训练时不匹配
  • 测试代码仅添加batch维度,摄像头处理额外添加了channel维度,若模型训练输入无channel维度,会引发形状不兼容

2. 冗余缩放破坏特征

摄像头循环中先将帧缩放到(200,200),预测函数又缩放到(64,64),两次缩放会丢失手势细节,尤其是边缘特征。

修正后的统一处理代码

import os
import cv2
import numpy as np
from keras.models import load_model

model = load_model("FINAL.h5")
class_mapping = train_images.class_indices
print(f"Class mapping: {class_mapping}")

def get_class_label(predictions, class_mapping):
    labels_mapping = {v: k for k, v in class_mapping.items()}
    predicted_labels = [labels_mapping[pred] for pred in predictions]
    print(predicted_labels)
    return predicted_labels

# 统一预处理逻辑,和测试图片保持完全一致
def predict_image_class(model, img_input, class_mapping, is_path=True):
    if is_path:
        img = cv2.imread(img_input)
        if img is None:
            print(f"Error: 无法加载图片 {img_input}")
            return ["Error"]
    else:
        img = img_input
    
    # 和测试代码完全对齐的步骤
    img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
    img = cv2.resize(img, (64, 64))
    # 若训练时未做归一化,注释下方代码
    # img = img.astype('float32') / 255.0
    img = np.expand_dims(img, axis=0)
    # 若模型输入为(64,64,1),取消注释下方代码
    # img = np.expand_dims(img, axis=-1)
    
    predictions = np.argmax(model.predict(img, verbose=0), axis=1)
    print(f"预测结果索引: {predictions}")
    predicted_labels = get_class_label(predictions, class_mapping)
    return predicted_labels

cap = cv2.VideoCapture(0)

while True:
    ret, frame = cap.read()
    if not ret:
        break

    # 直接传入原始帧,避免冗余缩放
    predicted_labels = predict_image_class(model, frame, class_mapping, is_path=False)
    
    # 仅为显示缩放帧,不影响预测逻辑
    display_frame = cv2.resize(frame, (400, 300))
    cv2.putText(display_frame, predicted_labels[0], (50, 50), cv2.FONT_HERSHEY_SIMPLEX, 1, (0, 255, 0), 2, cv2.LINE_AA)
    
    cv2.imshow('手语实时翻译', display_frame)
    
    if cv2.waitKey(1) & 0xFF == ord('q'):
        break

cap.release()
cv2.destroyAllWindows()

额外调试建议

  • 保存摄像头处理后的帧,用测试代码预测该图片,验证预处理是否正确
  • 调整摄像头光照,确保手势区域和训练集图片亮度、对比度一致
  • 打印model.input_shape,确认输入形状完全匹配

内容的提问来源于stack exchange,提问作者Daniel George

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.20 11:14:52