训练的TensorFlow人脸检测模型提取视频帧输出黑图,求排查
人脸检测模型导出后推理生成黑图问题排查与解决
我训练了一个TensorFlow人脸检测模型并导出为.h5文件,用另一个脚本加载模型从视频中提取检测到人脸的帧,但提取出的都是黑图,恳请技术指导。
训练代码
import tensorflow as tf import cv2 from tensorflow.keras import layers import os import numpy as np from PIL import Image from sklearn.model_selection import train_test_split from tensorflow.keras.applications.resnet50 import preprocess_input image_size = (224, 224) batch_size = 32 epochs = 10 def load_widerface_dataset(widerface_dir): images_dir = os.path.join(widerface_dir, 'WIDER_train', 'images') labels_file = os.path.join(widerface_dir, 'wider_face_split', 'wider_face_train_bbx_gt.txt') X = [] y = [] with open(labels_file, 'r') as f: lines = f.readlines() num_images = int(lines[1]) current_line = 2 for _ in range(num_images): image_path = os.path.join(images_dir, lines[current_line - 2].strip()) # 原代码错误:读取当前行的人脸数量,而非固定索引行 num_faces = int(lines[current_line].strip()) image = Image.open(image_path) image = image.resize((224, 224)) X.append(np.array(image)) # 取第一个人脸作为训练标签(适配当前单框输出模型) if num_faces > 0: face_line = lines[current_line + 1].strip().split(' ') face = [int(coord) for coord in face_line[:4]] # 框坐标归一化到0-1范围,匹配sigmoid输出 face = [ face[0]/image.size[0], face[1]/image.size[1], (face[0]+face[2])/image.size[0], (face[1]+face[3])/image.size[1] ] y.append(np.array(face)) else: y.append(np.array([0,0,0,0])) current_line += num_faces + 2 # 跳过人脸行和空行 X = np.array(X) y = np.array(y) return X, y def build_model(): model = tf.keras.models.Sequential([ tf.keras.layers.Conv2D(32, (3, 3), activation='relu', input_shape=(image_size[0], image_size[1], 3)), tf.keras.layers.MaxPooling2D((2, 2)), tf.keras.layers.Conv2D(64, (3, 3), activation='relu'), tf.keras.layers.MaxPooling2D((2, 2)), tf.keras.layers.Conv2D(64, (3, 3), activation='relu'), tf.keras.layers.Flatten(), tf.keras.layers.Dense(64, activation='relu'), tf.keras.layers.Dense(4, activation='sigmoid') ]) return model widerface_dir = 'C:/face_training/WIDER_train' X_train, y_train = load_widerface_dataset(widerface_dir) # 训练前归一化图像到0-1范围,与推理输入对齐 X_train = X_train / 255.0 model = build_model() model.compile(optimizer = 'adam', loss = 'mean_squared_error') model.fit(X_train, y_train, batch_size = batch_size, epochs = epochs) model.save('face_detection_model.h5')
原推理代码
import cv2 import os from tensorflow.keras.models import load_model import sys import numpy as np vidPath = "testclip2.mp4" model_path = 'face_detection_model.h5' model = load_model(model_path) test_tensorflow = 'C:/Users/user/Documents/PyCharmProjects/test_tensorflow' if not os.path.exists(test_tensorflow): os.makedirs(test_tensorflow) cap = cv2.VideoCapture(vidPath) currentFrame = 0 while (cap.isOpened()): ret, frame = cap.read() if ret == True: frame = cv2.resize(frame, (224, 224)) frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) frame = frame / 255.0 frames = [frame] for frame in frames: prediction = model.predict(frame.reshape(1, 224, 224, 3)) if prediction.any() > 0.5: cv2.imwrite(os.path.join(test_tensorflow, str(currentFrame) + '.jpg'), frame) currentFrame += 1 else: break cap.release() sys.exit(0)
问题排查与解决
1. 黑图直接原因:图像格式不匹配
cv2.imwrite要求输入图像是0-255范围的uint8类型,但原推理代码中把图像除以255转换成了0-1的float32类型,保存时会被自动截断为0,导致全黑。
修正方法:保存前将图像还原为0-255的uint8格式,同时把RGB转回BGR(cv2默认用BGR存储):
# 替换原保存代码 if prediction.any() > 0.5: save_frame = (frame * 255).astype(np.uint8) save_frame = cv2.cvtColor(save_frame, cv2.COLOR_RGB2BGR) cv2.imwrite(os.path.join(test_tensorflow, str(currentFrame) + '.jpg'), save_frame) currentFrame += 1
2. 训练代码核心错误:标签加载逻辑混乱
原load_widerface_dataset函数中,num_faces = int(lines[num_images])完全错误,应该读取当前行的人脸数量,否则会导致标签加载混乱,模型无法学到有效检测能力。
同时,原模型只能输出单个人脸框,而WIDERFACE数据集一张图可能有多个人脸,训练时需选择单个人脸(如第一个)作为标签,并将框坐标归一化到0-1范围(匹配sigmoid输出)。
3. 训练与推理输入分布不一致
原训练代码中X_train是0-255的数组,但推理输入是0-1的图像,分布不匹配会导致预测异常,训练前需将X_train除以255归一化。
修正后的完整推理代码
import cv2 import os from tensorflow.keras.models import load_model import sys import numpy as np vidPath = "testclip2.mp4" model_path = 'face_detection_model.h5' model = load_model(model_path) test_tensorflow = 'C:/Users/user/Documents/PyCharmProjects/test_tensorflow' if not os.path.exists(test_tensorflow): os.makedirs(test_tensorflow) cap = cv2.VideoCapture(vidPath) currentFrame = 0 while (cap.isOpened()): ret, frame = cap.read() if ret == True: original_frame = frame.copy() # 预处理用于模型预测 input_frame = cv2.resize(frame, (224, 224)) input_frame = cv2.cvtColor(input_frame, cv2.COLOR_BGR2RGB) input_frame = input_frame / 255.0 prediction = model.predict(input_frame.reshape(1, 224, 224, 3), verbose=0) # 更严谨的人脸判断逻辑 if (prediction > 0.1).all(): # 保存原始尺寸帧,避免缩放丢失细节 cv2.imwrite(os.path.join(test_tensorflow, f"{currentFrame}.jpg"), original_frame) currentFrame += 1 else: break cap.release() sys.exit(0)
内容的提问来源于stack exchange,提问作者jaytee
相关产品推荐
相关产品推荐

