You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在指定ROI区域实现YOLO目标检测与EasyOCR文本检测?

结合YOLO与EasyOCR实现车辆+车牌检测的ROI优化方案

你的代码核心问题在于没有利用YOLO检测出的车辆ROI缩小车牌检测范围,且车牌检测的预处理、参数设置存在不合理之处,导致ROI检测失效。以下是针对性的解决建议和修改后的代码:


关键优化点

  1. 绑定车辆ROI与车牌检测:仅在YOLO识别出的车辆区域内搜索车牌,避免全图误检,提升效率
  2. 优化车牌预处理流程:增加二值化、降噪操作,强化车牌特征,提升EasyOCR识别准确率
  3. 修复边界溢出问题:裁剪ROI时做边界校验,避免超出图像范围导致的无效处理
  4. 调整级联分类器参数:针对实际场景修改检测参数,降低误检概率

修改后的完整代码

import cv2
import numpy as np
import easyocr

# 加载YOLO模型
net = cv2.dnn.readNet('yolov4-tiny-custom_3000.weights', 'yolov4-tiny-custom.cfg')
classes = []
with open("obj.names", "r") as f:
    classes = [line.strip() for line in f.readlines()]
layer_names = net.getLayerNames()
output_layers = [layer_names[i - 1] for i in net.getUnconnectedOutLayers()]
colors = np.random.uniform(0, 255, size=(len(classes), 3))

# 初始化视频流
cap = cv2.VideoCapture('car1.mp4')

# 初始化OCR与车牌检测器
cascade_src = 'haarcascade_russian_plate_number.xml'
cascade = cv2.CascadeClassifier(cascade_src)
reader = easyocr.Reader(['en'], gpu=False)

while True:
    ret, frame = cap.read()
    if not ret:
        break  # 视频读取完毕退出循环
    height, width, channels = frame.shape

    # YOLO车辆检测流程
    blob = cv2.dnn.blobFromImage(frame, 0.00392, (416, 416), (0, 0, 0), True, crop=False)
    net.setInput(blob)
    outs = net.forward(output_layers)

    class_ids = []
    confidences = []
    boxes = []
    for out in outs:
        for detection in out:
            scores = detection[5:]
            class_id = np.argmax(scores)
            confidence = scores[class_id]
            if confidence > 0.5:
                # 计算车辆边界框坐标
                center_x = int(detection[0] * width)
                center_y = int(detection[1] * height)
                w = int(detection[2] * width)
                h = int(detection[3] * height)
                x = int(center_x - w / 2)
                y = int(center_y - h / 2)

                boxes.append([x, y, w, h])
                confidences.append(float(confidence))
                class_ids.append(class_id)

    # 非极大值抑制去除重复检测框
    indexes = cv2.dnn.NMSBoxes(boxes, confidences, 0.5, 0.4)
    for i in range(len(boxes)):
        if i in indexes:
            x_car, y_car, w_car, h_car = boxes[i]
            label = str(classes[class_ids[i]])
            color = colors[class_ids[i]]
            cv2.rectangle(frame, (x_car, y_car), (x_car + w_car, y_car + h_car), color, 2)
            cv2.putText(frame, label, (x_car, y_car + 30), cv2.FONT_HERSHEY_PLAIN, 3, color, 3)
            print("Jenis Mobil: " + label)

            # --- 核心修改:在车辆ROI内检测车牌 ---
            # 提取车辆区域ROI,避免超出图像边界
            car_roi = frame[max(0, y_car):min(y_car + h_car, height), max(0, x_car):min(x_car + w_car, width)]
            if car_roi.size == 0:
                continue  # 跳过无效ROI

            gray_car = cv2.cvtColor(car_roi, cv2.COLOR_BGR2GRAY)
            # 调整级联分类器参数,适配场景
            plates = cascade.detectMultiScale(gray_car, scaleFactor=1.05, minNeighbors=3, minSize=(30, 10))

            for x_plate, y_plate, w_plate, h_plate in plates:
                # 裁剪车牌ROI并增加边界偏移
                offset = 2
                plate_roi = car_roi[max(0, y_plate - offset):min(y_plate + h_plate + offset, car_roi.shape[0]),
                                    max(0, x_plate - offset):min(x_plate + w_plate + offset, car_roi.shape[1])]
                
                # 预处理:二值化+降噪,强化车牌特征
                gray_plate = cv2.cvtColor(plate_roi, cv2.COLOR_BGR2GRAY)
                _, thresh_plate = cv2.threshold(gray_plate, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
                blur_plate = cv2.GaussianBlur(thresh_plate, (3, 3), 0)

                # EasyOCR识别预处理后的车牌
                result = reader.readtext(blur_plate, min_confidence=0.7)
                for detek in result:
                    text = detek[1]
                    # 将车牌坐标转换回原图坐标系
                    orig_x = x_car + x_plate
                    orig_y = y_car + y_plate
                    # 绘制车牌框与识别结果
                    cv2.rectangle(frame, (orig_x, orig_y), (orig_x + w_plate, orig_y + h_plate), (60, 60, 255), 2)
                    cv2.rectangle(frame, (orig_x - 1, orig_y - 40), (orig_x + w_plate + 1, orig_y), (60, 60, 255), -1)
                    cv2.putText(frame, text, (orig_x, orig_y - 10), cv2.FONT_HERSHEY_SIMPLEX, 0.9, (255, 255, 255), 2)
                    print("Nomor Kendaran: " + text)

    cv2.imshow("Detection", frame)
    key = cv2.waitKey(1)
    if key == 27:
        break

cap.release()
cv2.destroyAllWindows()

额外提示

  • 如果俄罗斯车牌级联分类器效果不佳,建议替换为针对你所在地区车牌训练的分类器,或直接用YOLO训练专属车牌检测模型,准确率会显著提升
  • 可根据视频帧率增加帧跳过逻辑(如每2帧处理一次),平衡检测精度与处理速度

内容的提问来源于stack exchange,提问作者Arif Fadillah

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.06 02:01:03