You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

训练的TensorFlow人脸检测模型提取视频帧输出黑图,求排查

人脸检测模型导出后推理生成黑图问题排查与解决

我训练了一个TensorFlow人脸检测模型并导出为.h5文件,用另一个脚本加载模型从视频中提取检测到人脸的帧,但提取出的都是黑图,恳请技术指导。

训练代码

import tensorflow as tf
import cv2
from tensorflow.keras import layers
import os
import numpy as np
from PIL import Image
from sklearn.model_selection import train_test_split
from tensorflow.keras.applications.resnet50 import preprocess_input

image_size = (224, 224)
batch_size = 32
epochs = 10

def load_widerface_dataset(widerface_dir):
    images_dir = os.path.join(widerface_dir, 'WIDER_train', 'images')
    labels_file = os.path.join(widerface_dir, 'wider_face_split', 'wider_face_train_bbx_gt.txt')

    X = []
    y = []

    with open(labels_file, 'r') as f:
        lines = f.readlines()

    num_images = int(lines[1])
    current_line = 2

    for _ in range(num_images):
        image_path = os.path.join(images_dir, lines[current_line - 2].strip())
        # 原代码错误:读取当前行的人脸数量,而非固定索引行
        num_faces = int(lines[current_line].strip())

        image = Image.open(image_path)
        image = image.resize((224, 224))
        X.append(np.array(image))

        # 取第一个人脸作为训练标签(适配当前单框输出模型)
        if num_faces > 0:
            face_line = lines[current_line + 1].strip().split(' ')
            face = [int(coord) for coord in face_line[:4]]
            # 框坐标归一化到0-1范围,匹配sigmoid输出
            face = [
                face[0]/image.size[0],
                face[1]/image.size[1],
                (face[0]+face[2])/image.size[0],
                (face[1]+face[3])/image.size[1]
            ]
            y.append(np.array(face))
        else:
            y.append(np.array([0,0,0,0]))

        current_line += num_faces + 2  # 跳过人脸行和空行

    X = np.array(X)
    y = np.array(y)

    return X, y

def build_model():
    model = tf.keras.models.Sequential([
        tf.keras.layers.Conv2D(32, (3, 3), activation='relu',
                               input_shape=(image_size[0], image_size[1], 3)),
        tf.keras.layers.MaxPooling2D((2, 2)),
        tf.keras.layers.Conv2D(64, (3, 3), activation='relu'),
        tf.keras.layers.MaxPooling2D((2, 2)),
        tf.keras.layers.Conv2D(64, (3, 3), activation='relu'),
        tf.keras.layers.Flatten(),
        tf.keras.layers.Dense(64, activation='relu'),
        tf.keras.layers.Dense(4, activation='sigmoid')
    ])

    return model

widerface_dir = 'C:/face_training/WIDER_train'
X_train, y_train = load_widerface_dataset(widerface_dir)
# 训练前归一化图像到0-1范围,与推理输入对齐
X_train = X_train / 255.0
model = build_model()
model.compile(optimizer = 'adam', loss = 'mean_squared_error')
model.fit(X_train, y_train, batch_size = batch_size, epochs = epochs)
model.save('face_detection_model.h5')

原推理代码

import cv2
import os
from tensorflow.keras.models import load_model
import sys
import numpy as np

vidPath = "testclip2.mp4"
model_path = 'face_detection_model.h5'
model = load_model(model_path)

test_tensorflow = 'C:/Users/user/Documents/PyCharmProjects/test_tensorflow'
if not os.path.exists(test_tensorflow):
    os.makedirs(test_tensorflow)

cap = cv2.VideoCapture(vidPath)

currentFrame = 0
while (cap.isOpened()):
    ret, frame = cap.read()

    if ret == True:
        frame = cv2.resize(frame, (224, 224))
        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
        frame = frame / 255.0
        frames = [frame]

        for frame in frames:
            prediction = model.predict(frame.reshape(1, 224, 224, 3))

            if prediction.any() > 0.5:
                cv2.imwrite(os.path.join(test_tensorflow, str(currentFrame) + '.jpg'), frame)
                currentFrame += 1

    else:
        break

cap.release()
sys.exit(0)

问题排查与解决

1. 黑图直接原因:图像格式不匹配

cv2.imwrite要求输入图像是0-255范围的uint8类型,但原推理代码中把图像除以255转换成了0-1的float32类型,保存时会被自动截断为0,导致全黑。

修正方法:保存前将图像还原为0-255的uint8格式,同时把RGB转回BGR(cv2默认用BGR存储):

# 替换原保存代码
if prediction.any() > 0.5:
    save_frame = (frame * 255).astype(np.uint8)
    save_frame = cv2.cvtColor(save_frame, cv2.COLOR_RGB2BGR)
    cv2.imwrite(os.path.join(test_tensorflow, str(currentFrame) + '.jpg'), save_frame)
    currentFrame += 1

2. 训练代码核心错误:标签加载逻辑混乱

原load_widerface_dataset函数中,num_faces = int(lines[num_images])完全错误,应该读取当前行的人脸数量,否则会导致标签加载混乱,模型无法学到有效检测能力。

同时,原模型只能输出单个人脸框,而WIDERFACE数据集一张图可能有多个人脸,训练时需选择单个人脸(如第一个)作为标签,并将框坐标归一化到0-1范围(匹配sigmoid输出)。

3. 训练与推理输入分布不一致

原训练代码中X_train是0-255的数组,但推理输入是0-1的图像,分布不匹配会导致预测异常,训练前需将X_train除以255归一化。

修正后的完整推理代码

import cv2
import os
from tensorflow.keras.models import load_model
import sys
import numpy as np

vidPath = "testclip2.mp4"
model_path = 'face_detection_model.h5'
model = load_model(model_path)

test_tensorflow = 'C:/Users/user/Documents/PyCharmProjects/test_tensorflow'
if not os.path.exists(test_tensorflow):
    os.makedirs(test_tensorflow)

cap = cv2.VideoCapture(vidPath)

currentFrame = 0
while (cap.isOpened()):
    ret, frame = cap.read()

    if ret == True:
        original_frame = frame.copy()
        # 预处理用于模型预测
        input_frame = cv2.resize(frame, (224, 224))
        input_frame = cv2.cvtColor(input_frame, cv2.COLOR_BGR2RGB)
        input_frame = input_frame / 255.0

        prediction = model.predict(input_frame.reshape(1, 224, 224, 3), verbose=0)
        # 更严谨的人脸判断逻辑
        if (prediction > 0.1).all():
            # 保存原始尺寸帧,避免缩放丢失细节
            cv2.imwrite(os.path.join(test_tensorflow, f"{currentFrame}.jpg"), original_frame)
            currentFrame += 1

    else:
        break

cap.release()
sys.exit(0)

内容的提问来源于stack exchange,提问作者jaytee

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.21 12:07:02