Flask部署LSTM手语识别项目:摄像头卡顿且动作无检测
手语识别Flask应用:卡顿+动作识别失效问题排查与修复
问题根源拆解
- 语音合成阻塞线程:
t2s.runAndWait()是同步阻塞操作,会直接暂停视频帧生成流程,既导致应用卡顿,也打断识别逻辑的连续执行。 - 变量作用域混乱:全局
predictions未在帧生成函数内初始化,多次运行会累积无效历史数据;局部sequence/sentence每次请求都重置,导致需要重新积累30帧才能触发识别,容易让人误以为识别功能失效。 - 单线程承载计算密集型任务:Mediapipe关键点提取、LSTM模型预测都在视频帧生成主线程同步执行,单线程扛不住实时处理的计算量,引发卡顿。
- 预测一致性判断逻辑不可靠:用
np.unique(predictions[-10:])[0]判断连续预测的方式有bug——如果前10次预测存在多个类别,会直接取第一个唯一值而非多数结果,导致识别触发条件极难满足。
针对性修复方案
1. 异步处理语音合成
把语音合成放到独立线程执行,彻底避免阻塞视频流:
def speak_word(word): t2s.say(word) t2s.runAndWait() # 在generate_frames的识别成功处替换原语音代码: threading.Thread(target=speak_word, args=(new_word,), daemon=True).start()
2. 修正变量作用域
将predictions移到generate_frames函数内初始化,确保每次视频会话的预测历史独立:
def generate_frames(): sequence = [] sentence = [] predictions = [] # 函数内初始化,避免全局污染 threshold = 0.5 # ... 后续逻辑 ...
3. 优化预测一致性判断
手动实现众数统计替换原有的np.unique逻辑,确保连续预测的稳定性:
def get_mode(arr): counts = np.bincount(arr) return np.argmax(counts) # 在模型预测后替换原判断逻辑: if len(predictions) >= 10: most_common = get_mode(predictions[-10:]) if most_common == np.argmax(res) and res[np.argmax(res)] > threshold: # ... 后续添加句子和语音逻辑 ...
4. 减轻主线程计算压力
- 降低视频分辨率,减少Mediapipe处理的像素量:
def generate_frames(): cap = cv2.VideoCapture(0) # 设置低分辨率 cap.set(cv2.CAP_PROP_FRAME_WIDTH, 640) cap.set(cv2.CAP_PROP_FRAME_HEIGHT, 480) # ... 后续逻辑 ...
- 启用TensorFlow XLA加速,提升模型推理速度:
import tensorflow as tf tf.config.optimizer.set_jit(True) model = load_model('action.h5')
5. 修复线程安全问题
将视频捕获和Mediapipe模型移到generate_frames函数内初始化,避免全局资源在多线程下的冲突:
def generate_frames(): cap = cv2.VideoCapture(0) holistic = mp_holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.5) try: # ... 帧处理逻辑 ... finally: cap.release() holistic.close() # 及时释放资源
6. 添加调试日志
在关键节点打印日志,方便定位识别是否触发:
# 模型预测后打印 res = model.predict(np.expand_dims(sequence, axis=0), verbose=0)[0] pred_idx = np.argmax(res) print(f"预测类别:{actions[pred_idx]},概率:{res[pred_idx]:.2f}") # 识别成功时打印 print(f"识别成功,添加到句子:{new_word}")
完整修改后的核心代码片段
from flask import Flask, render_template, Response import cv2 import pyttsx3 import numpy as np import mediapipe as mp import threading import tensorflow as tf # 启用XLA加速推理 tf.config.optimizer.set_jit(True) app = Flask(__name__) from tensorflow.keras.models import load_model model = load_model('action.h5') mp_holistic = mp.solutions.holistic mp_drawing = mp.solutions.drawing_utils actions = np.array(['Hello','I am','Affan','Thanks', 'i love you','Fever','See you', 'God']) t2s = pyttsx3.init() def mediapipe_detection(image, model): image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) image.flags.writeable = False results = model.process(image) image.flags.writeable = True image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR) return image, results def extract_keypoints(results): lh = np.array([[res.x, res.y, res.z] for res in results.left_hand_landmarks.landmark]).flatten() if results.left_hand_landmarks else np.zeros(21*3) rh = np.array([[res.x, res.y, res.z] for res in results.right_hand_landmarks.landmark]).flatten() if results.right_hand_landmarks else np.zeros(21*3) return np.concatenate([lh, rh]) def draw_styled_landmarks(image, results): if results.left_hand_landmarks: mp_drawing.draw_landmarks(image, results.left_hand_landmarks, mp_holistic.HAND_CONNECTIONS, mp_drawing.DrawingSpec(color=(100, 100, 100), thickness=2, circle_radius=4), mp_drawing.DrawingSpec(color=(100, 100, 100), thickness=2, circle_radius=2) ) if results.right_hand_landmarks: mp_drawing.draw_landmarks(image, results.right_hand_landmarks, mp_holistic.HAND_CONNECTIONS, mp_drawing.DrawingSpec(color=(200, 200,200), thickness=2, circle_radius=4), mp_drawing.DrawingSpec(color=(200, 200, 200), thickness=2, circle_radius=2) ) def speak_word(word): t2s.say(word) t2s.runAndWait() def get_mode(arr): counts = np.bincount(arr) return np.argmax(counts) def generate_frames(): cap = cv2.VideoCapture(0) cap.set(cv2.CAP_PROP_FRAME_WIDTH, 640) cap.set(cv2.CAP_PROP_FRAME_HEIGHT, 480) holistic = mp_holistic.Holistic(min_detection_confidence=0.5, min_tracking_confidence=0.5) sequence = [] sentence = [] predictions = [] threshold = 0.5 try: while True: ret, frame = cap.read() if not ret: break image, results = mediapipe_detection(frame, holistic) draw_styled_landmarks(image, results) keypoints = extract_keypoints(results) sequence.append(keypoints) sequence = sequence[-30:] if len(sequence) == 30: res = model.predict(np.expand_dims(sequence, axis=0), verbose=0)[0] pred_idx = np.argmax(res) predictions.append(pred_idx) print(f"预测类别:{actions[pred_idx]},概率:{res[pred_idx]:.2f}") if len(predictions) >= 10: most_common = get_mode(predictions[-10:]) if most_common == pred_idx and res[pred_idx] > threshold: if len(sentence) == 0 or actions[pred_idx] != sentence[-1]: sentence.append(actions[pred_idx]) new_word = actions[pred_idx] print(f"识别成功:{new_word}") threading.Thread(target=speak_word, args=(new_word,), daemon=True).start() if len(sentence) > 5: sentence = sentence[-5:] ret, buffer = cv2.imencode('.jpg', image) frame = buffer.tobytes() yield (b'--frame\r\n' b'Content-Type: image/jpeg\r\n\r\n' + frame + b'\r\n') finally: cap.release() holistic.close() @app.route('/') def index(): return render_template('index.html') @app.route('/video_feed') def video_feed(): return Response(generate_frames(), mimetype='multipart/x-mixed-replace; boundary=frame') if __name__ == "__main__": app.run(debug=True)
内容的提问来源于stack exchange,提问作者Affan M N M
相关产品推荐
相关产品推荐

