You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Python OpenCV实现直播电视宏块检测及文字误检区分方法问询

直播电视画面宏块像素化检测实现

我正在开发一款脚本,用于检测外接摄像头采集的直播电视画面中的像素化问题,目前使用包含2处像素化片段的直播电视短素材开展测试。
初始阶段已经可以过滤掉视频中大部分噪声并检出像素化区域,但所使用的内核会将高亮白色文字误识别为像素化区域,需要实现真实像素化区域与文字区域的有效区分。

初始实现代码

import cv2
import numpy as np

cap = cv2.VideoCapture("./hgtv_short.ts")

while True:
    success, image = cap.read()

    gray = cv2.cvtColor(src=image, code=cv2.COLOR_BGR2GRAY)

    sharpen_kernel = np.array([[.4, .4], [-2.25, -2.25], [.4, .4]])
    sharpen = cv2.filter2D(src=gray, ddepth=-1, kernel=sharpen_kernel)
    sharpe = sharpen + 128

    canny = cv2.Canny(image=sharpe, threshold1=245, threshold2=255, edges=1, apertureSize=3, L2gradient=True)
    white = np.where(canny != [0])
    coordinates = zip(white[1], white[0])
    for p in coordinates:
        cv2.circle(canny, p, 30, (200, 0, 0), 2)

    cv2.imshow('image', image)
    cv2.imshow('edges', canny)

    cv2.waitKey(1)

问题明确

需要检测的具体像素化类型为视频压缩带来的宏块效应,实际测试中已成功检出宏块区域,但同步误检出了画面中的白色文字区域,需要优化逻辑做二分类区分。

优化方案落地

经过多次试错,最终确定使用参考模型预测图像是否存在宏块、像素化、伪影等问题的方案效果最优,选择HOG(方向梯度直方图)描述子生成特征向量,搭配SVM(支持向量机)分类算法训练检测模型,具体实现如下:

正负样本特征采集

def pos_train_set(self):
    print("Starting to Gather Positive Photos")
    for pos_file in glob.iglob(os.path.join(self.base_path, "Bad_Images", "*.jpg")):
        pos_img = cv2.imread(pos_file, 1)
        pos_img = cv2.resize(pos_img, self.winSize, interpolation=cv2.CV_32F)
        pos_des = self.hog.compute(pos_img)
        pos_des = cv2.normalize(pos_des, None)
        self.labels.append(1)
        self.training_data.append(pos_des)
    print("Gathered Positive Photos")

def neg_train_set(self):
    print("Starting to Gather Negative Photos")
    for neg_file in glob.iglob(os.path.join(self.base_path, "Good_Images", "*.jpg")):
        neg_img = cv2.imread(neg_file, 1)
        neg_img = cv2.resize(neg_img, self.winSize, interpolation=cv2.CV_32F)
        neg_des = self.hog.compute(neg_img)
        neg_des = cv2.normalize(neg_des, None)
        self.labels.append(0)
        self.training_data.append(neg_des)
    print("Gathered Negative Photos")

SVM模型训练

def train_set(self):
    print("Starting to Convert")
    td = np.float32(self.training_data)
    lab = np.array(self.labels)

    print("Converted List")
    print("Starting Shuffle")
    rand = np.random.RandomState(10)
    shuffle = rand.permutation(len(td))
    td = td[shuffle]
    lab = lab[shuffle]

    print("Shuffled List")
    print("Starting SVM")

    svm = cv2.ml.SVM_create()
    svm.setType(cv2.ml.SVM_C_SVC)
    # Exponential Chi2 kernel, similar to the RBF kernel: K(xi,xj)=e−γχ2(xi,xj),χ2(xi,xj)=(xi−xj)2/(xi+xj),γ>0.
    svm.setKernel(cv2.ml.SVM_CHI2)
    svm.setTermCriteria((cv2.TERM_CRITERIA_MAX_ITER, 100, 1e-6))
    svm.setGamma(5.383)
    svm.setC(2.67)
    print("Starting Training")

    svm.train(td, cv2.ml.ROW_SAMPLE, lab)

    print("Saving to .yml")

    svm.save(os.path.join(self.base_path, "svm_model.yml"))

推理预测逻辑

加载训练好的SVM模型,结合初始的内核和边缘检测逻辑,判断图像是否存在异常:

def predict(self):
    svm = cv2.ml.SVM_load("./svm_model.yml")

    for file in self.files:
        os.mkdir(os.path.join(self.base_path, "1_Frames", os.path.basename(file)))
        print(f"Starting predict on {file}")
        cap = cv2.VideoCapture(file)
        while cap.isOpened():
            success, image = cap.read(1)
            if success:
                img = cv2.resize(image, self.winSize, interpolation=cv2.CV_32F)
                test_data = self.hog.compute(img)
                test_data = cv2.normalize(test_data, None)
                test_data = np.float32(test_data)
                test_data = np.transpose(test_data)

                if not np.any(test_data):
                    print("Invalid Dimension")
                    success, image = cap.read(1)
                    print(f"New Frame {success}")
                else:
                    response = svm.predict(test_data)[1]
                    if response == 1:
                        gray = cv2.cvtColor(src=image, code=cv2.COLOR_BGR2GRAY)
                        sharpen_kernel = np.array([[.4, .4], [-2.25, -2.25], [.4, .4]])
                        sharpen = cv2.filter2D(src=gray, ddepth=-1, kernel=sharpen_kernel)
                        sharpe = sharpen + 128

                        canny = cv2.Canny(image=sharpe, threshold1=245, threshold2=255, edges=1, apertureSize=3, L2gradient=True)
                        white = np.where(canny != [0])
                        if not len(white[0]) == 0:
                            cv2.imwrite(os.path.join(self.base_path, '1_Frames', os.path.basename(file), f'found_{self.x}.jpg'), image)
                            success, image = cap.read(1)
                            self.x += 1
                        else:
                            success, image = cap.read(1)
                            pass
                    else:
                        cv2.imwrite(os.path.join(self.base_path, '0_Frames', f'found_{self.y}.jpg'), image)
                        success, image = cap.read(1)
                        self.y += 1
            else:
                break

        cap.release()
        cv2.destroyAllWindows()

该方案目前运行效果较好,也可以为其他有图像异常检测需求的开发者提供参考。

内容的提问来源于stack exchange,提问作者Ryan Burch

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.10.03 09:06:01