如何在不同尺寸大图中定位目标图像?PyAutoGUI与OpenCV均失效
问题描述
我尝试使用pyautogui.locateOnScreen(btn, confidence=0.8, grayscale=True)定位图像,但当搜索图像尺寸不同时无法找到目标。
我的需求是定位屏幕上的确认按钮:截取了网站A上的小尺寸按钮,想在网站B上定位它,但未能成功。需要在不同网站上检测该按钮的位置,除尺寸外,按钮的颜色、形状、文字等其他特征均一致。
我还尝试了cv2.matchTemplate(scr_img, search_image, cv2.TM_CCOEFF_NORMED),但同样无效。请问还有其他解决方法吗?
已尝试的代码
使用pyautogui.locateOnScreen的代码
from PIL import Image confirm_btn_img = Image.open("./img/confirm_btn.png") try: l, t, w, h = pyautogui.locateOnScreen(confirm_btn_img, confidence=0.8, grayscale=True) except Exception as e: print("Not Found")
使用cv2.matchTemplate的代码
import cv2 import numpy as np confirm_btn_img2 = cv2.imread("./img/confirm_btn.png", 0) big_image = cv2.cvtColor(np.array(pyautogui.screenshot()), cv2.COLOR_RGB2GRAY) res = cv2.matchTemplate(big_image, confirm_btn_img2, cv2.TM_CCOEFF_NORMED) threshold = 0.8 loc = np.where(res >= threshold) print("loc====", loc) for pt in zip(*loc[::-1]): cv2.rectangle(big_image, pt, (pt[0] + confirm_btn_img2.shape[1], pt[1] + confirm_btn_img2.shape[0]), (0, 0, 255), 2) cv2.imshow('Detected', big_image) cv2.waitKey(0) cv2.destroyAllWindows()
示例图片
- 可成功定位的示例:

- 无法定位的示例:

- 目标图像:

可成功运行的示例代码
import cv2 import numpy as np confirm_btn_img2 = cv2.imread("./test/ok.png", 0) big_image = cv2.imread("./test/can_found_tpl.png", 0) res = cv2.matchTemplate(big_image, confirm_btn_img2, cv2.TM_CCOEFF_NORMED) threshold = 0.8 loc = np.where(res >= threshold) print("loc====", loc) for pt in zip(*loc[::-1]): cv2.rectangle(big_image, pt, (pt[0] + confirm_btn_img2.shape[1], pt[1] + confirm_btn_img2.shape[0]), (0, 0, 255), 2) cv2.imshow('Detected', big_image) cv2.waitKey(0) cv2.destroyAllWindows()
解决方法
针对不同尺寸的相同特征按钮检测,以下几种方案可以尝试:
1. 多尺度模板匹配(最直接的解决方式)
cv2.matchTemplate仅支持同尺寸匹配,需对目标图像进行多尺度缩放后逐一匹配,找到最符合的结果。
示例代码:
import cv2 import numpy as np import pyautogui template = cv2.imread("./img/confirm_btn.png", 0) h, w = template.shape[:2] screen = cv2.cvtColor(np.array(pyautogui.screenshot()), cv2.COLOR_RGB2GRAY) threshold = 0.8 found = None # 遍历不同缩放比例,范围可根据实际情况调整 for scale in np.linspace(0.5, 2.0, 20)[::-1]: # 缩放屏幕图像 resized = cv2.resize(screen, (int(screen.shape[1] * scale), int(screen.shape[0] * scale))) r = screen.shape[1] / float(resized.shape[1]) # 如果缩放后的图像比模板还小,停止循环 if resized.shape[0] < h or resized.shape[1] < w: break # 模板匹配 result = cv2.matchTemplate(resized, template, cv2.TM_CCOEFF_NORMED) loc = np.where(result >= threshold) # 记录最佳匹配结果 for (x, y) in zip(*loc[::-1]): if found is None or result[y, x] > found[0]: found = (result[y, x], (int(x * r), int(y * r), int(w * r), int(h * r))) if found is not None: print(f"找到按钮位置:{found[1]}") # 绘制矩形标记 cv2.rectangle(screen, (found[1][0], found[1][1]), (found[1][0] + found[1][2], found[1][1] + found[1][3]), (0, 0, 255), 2) cv2.imshow('Detected', screen) cv2.waitKey(0) cv2.destroyAllWindows() else: print("未找到目标")
2. 基于特征点的匹配(ORB)
这类方法不依赖图像尺寸,通过提取图像关键特征点匹配,适合形状、纹理一致但尺寸不同的场景。ORB为OpenCV自带,无需额外安装。
示例代码:
import cv2 import numpy as np import pyautogui template = cv2.imread("./img/confirm_btn.png", 0) screen = cv2.cvtColor(np.array(pyautogui.screenshot()), cv2.COLOR_RGB2GRAY) # 初始化ORB检测器 orb = cv2.ORB_create() # 提取特征点和描述符 kp1, des1 = orb.detectAndCompute(template, None) kp2, des2 = orb.detectAndCompute(screen, None) # 暴力匹配 bf = cv2.BFMatcher(cv2.NORM_HAMMING, crossCheck=True) matches = bf.match(des1, des2) # 按匹配度排序 matches = sorted(matches, key=lambda x: x.distance) # 筛选匹配点,取前10个最佳匹配 good_matches = matches[:10] if len(good_matches) > 5: # 匹配点数量足够则判定找到目标 src_pts = np.float32([kp1[m.queryIdx].pt for m in good_matches]).reshape(-1, 1, 2) dst_pts = np.float32([kp2[m.trainIdx].pt for m in good_matches]).reshape(-1, 1, 2) # 计算单应性矩阵,获取模板在屏幕中的位置 M, mask = cv2.findHomography(src_pts, dst_pts, cv2.RANSAC, 5.0) h, w = template.shape pts = np.float32([[0, 0], [0, h-1], [w-1, h-1], [w-1, 0]]).reshape(-1, 1, 2) if M is not None: dst = cv2.perspectiveTransform(pts, M) screen = cv2.polylines(screen, [np.int32(dst)], True, (0, 0, 255), 2) print(f"按钮位置:{np.int32(dst)}") cv2.imshow('Detected', screen) cv2.waitKey(0) cv2.destroyAllWindows() else: print("未找到目标")
3. OCR文字识别(针对带固定文字的按钮)
如果按钮文字固定(如“确认”“OK”),可通过OCR识别文字位置定位按钮,推荐使用pytesseract。
步骤:
- 安装依赖:
pip install pytesseract pillow opencv-python,并安装Tesseract OCR引擎(需配置系统路径) - 示例代码:
import cv2 import pytesseract import pyautogui from PIL import Image # 设置Tesseract路径(若未在系统PATH中) # pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe' screen = pyautogui.screenshot() screen_cv = cv2.cvtColor(np.array(screen), cv2.COLOR_RGB2BGR) gray = cv2.cvtColor(screen_cv, cv2.COLOR_BGR2GRAY) # 预处理图像(提高识别准确率) gray = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY | cv2.THRESH_OTSU)[1] # 识别文字及位置 data = pytesseract.image_to_data(gray, output_type=pytesseract.Output.DICT) target_text = "确认" # 替换为按钮上的文字 for i in range(len(data['text'])): if data['text'][i].strip() == target_text: x = data['left'][i] y = data['top'][i] w = data['width'][i] h = data['height'][i] print(f"找到按钮位置:({x}, {y}, {w}, {h})") cv2.rectangle(screen_cv, (x, y), (x + w, y + h), (0, 0, 255), 2) cv2.imshow('Detected', screen_cv) cv2.waitKey(0) cv2.destroyAllWindows() break else: print("未找到目标")
4. 调整pyautogui匹配参数(辅助优化)
若坚持使用pyautogui,可尝试:
- 降低
confidence阈值(如0.6-0.7),但可能增加误匹配 - 关闭
grayscale,保留颜色信息(若颜色特征稳定) - 生成多个不同尺寸的模板图片,逐一尝试
locateOnScreen
内容的提问来源于stack exchange,提问作者afraid.jpg
相关产品推荐
相关产品推荐

