You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

OpenCV Python拼接黑背景异尺寸图像:仅拼接右下区域问题求助

问题

开发的无人机航拍图像拼接程序采用两两拼接模式(已拼接结果作为查询图,与下一张图像拼接),存在以下问题:

  • 仅能拼接图像的右侧和底部区域,无法处理待拼接图像包含基底图没有的顶部、左侧区域的情况
  • 用Kaggle开源无人机图像测试时,仅保留训练图右侧部分,丢失顶部区域
  • 用网络裁剪图像测试时,第一行拼接正常,与第二行首图拼接也没问题,但拼接第二行第二图时,虽有特征匹配却无有效拼接结果,怀疑是查询图中的黑背景导致
  • 参考方案添加掩码忽略黑区域,但效果未改善

核心代码:

feature_extraction_algo = 'surf'

feature_to_match = 'knn'

train_photo = cv2.imread('./'  + 'train.jpg')

# OpenCV defines the color channel in the order BGR 
# Hence converting to RGB for Matplotlib
train_photo = cv2.cvtColor(train_photo,cv2.COLOR_BGR2RGB)

# converting to grayscale
train_photo_gray = cv2.cvtColor(train_photo, cv2.COLOR_RGB2GRAY)

# Do the same for the query image 
query_photo = cv2.imread('./'  + 'query.jpg')
query_photo = cv2.cvtColor(query_photo,cv2.COLOR_BGR2RGB)
query_photo_gray = cv2.cvtColor(query_photo, cv2.COLOR_RGB2GRAY)

def select_descriptor_methods(image, method=None):    
    
    assert method is not None, "Please define a feature descriptor method. accepted Values are: 'sift', 'surf', 'brisk', 'orb'"

    if method == 'sift':
        descriptor = cv2.SIFT_create()
    elif method == 'surf':
        descriptor = cv2.xfeatures2d.SURF_create()
    elif method == 'brisk':
        descriptor = cv2.BRISK_create()
    elif method == 'orb':
        descriptor = cv2.ORB_create()
        
    (keypoints, features) = descriptor.detectAndCompute(image, None)
    
    return (keypoints, features)

keypoints_train_img, features_train_img = select_descriptor_methods(train_photo_gray, method=feature_extraction_algo)

keypoints_query_img, features_query_img = select_descriptor_methods(query_photo_gray, method=feature_extraction_algo)

def create_matching_object(method,crossCheck):
    
    # For BF matcher, first we have to create the BFMatcher object using cv2.BFMatcher(). 
    # It takes two optional params. 
    # normType - It specifies the distance measurement
    # crossCheck - which is false by default. If it is true, Matcher returns only those matches 
    # with value (i,j) such that i-th descriptor in set A has j-th descriptor in set B as the best match 
    # and vice-versa. 
    if method == 'sift' or method == 'surf':
        bf = cv2.BFMatcher(cv2.NORM_L2, crossCheck=crossCheck)
    elif method == 'orb' or method == 'brisk':
        bf = cv2.BFMatcher(cv2.NORM_HAMMING, crossCheck=crossCheck)
    return bf

def key_points_matching(features_train_img, features_query_img, method):
    bf = create_matching_object(method, crossCheck=True)
        
    # Match descriptors.
    best_matches = bf.match(features_train_img,features_query_img)
    
    # Sort the features in order of distance.
    # The points with small distance (more similarity) are ordered first in the vector
    rawMatches = sorted(best_matches, key = lambda x:x.distance)
    print("Raw matches with Brute force):", len(rawMatches))
    return rawMatches

def key_points_matching_KNN(features_train_img, features_query_img, ratio, method):
    bf = create_matching_object(method, crossCheck=False)
    # compute the raw matches and initialize the list of actual matches
    rawMatches = bf.knnMatch(features_train_img, features_query_img, k=2)
    print("Raw matches (knn):", len(rawMatches))
    matches = []

    # loop over the raw matches
    for m,n in rawMatches:
        # ensure the distance is within a certain ratio of each
        # other (i.e. Lowe's ratio test)
        if m.distance < n.distance * ratio:
            matches.append(m)
    return matches

if feature_to_match == 'bf':
    matches = key_points_matching(features_train_img, features_query_img, method=feature_extraction_algo)
    
    mapped_features_image = cv2.drawMatches(train_photo,keypoints_train_img,query_photo,keypoints_query_img,matches[:100],
                           None,flags=cv2.DrawMatchesFlags_NOT_DRAW_SINGLE_POINTS)

# Now for cross checking draw the feature-mapping lines also with KNN
elif feature_to_match == 'knn':
    matches = key_points_matching_KNN(features_train_img, features_query_img, ratio=0.75, method=feature_extraction_algo)
    
    mapped_features_image = cv2.drawMatches(train_photo, keypoints_train_img, query_photo, keypoints_query_img, np.random.choice(matches,100),
                           None,flags=cv2.DrawMatchesFlags_NOT_DRAW_SINGLE_POINTS)

def homography_stitching(keypoints_train_img, keypoints_query_img, matches, reprojThresh):
    """ converting the keypoints to numpy arrays before passing them for calculating Homography Matrix.
    
    Because we are supposed to pass 2 arrays of coordinates to cv2.findHomography, as in I have these points in image-1, and I have points in image-2, so now what is the homography matrix to transform the points from image 1 to image 2
    """
    keypoints_train_img = np.float32([keypoint.pt for keypoint in keypoints_train_img])
    keypoints_query_img = np.float32([keypoint.pt for keypoint in keypoints_query_img])
    
    ''' For findHomography() - I need to have an assumption of a minimum of correspondence points that are present between the 2 images. Here, I am assuming that Minimum Match Count to be 4 '''
    if len(matches) > 4:
        # construct the two sets of points
        points_train = np.float32([keypoints_train_img[m.queryIdx] for m in matches])
        points_query = np.float32([keypoints_query_img[m.trainIdx] for m in matches])
        
        # Calculate the homography between the sets of points
        (H, status) = cv2.findHomography(points_train, points_query, cv2.RANSAC, reprojThresh)

        return (matches, H, status)
    else:
        return None

M = homography_stitching(keypoints_train_img, keypoints_query_img, matches, reprojThresh=4)

if M is None:
    print("Error!")

(matches, Homography_Matrix, status) = M

width = query_photo.shape[1] + train_photo.shape[1]
height = query_photo.shape[0] + train_photo.shape[0]

result = cv2.warpPerspective(train_photo, Homography_Matrix,  (width, height))

result[0:query_photo.shape[0], 0:query_photo.shape[1]] = query_photo

img = result

gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)

# threshold 
thresh = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY+cv2.THRESH_OTSU)[1]
hh, ww = thresh.shape
print(thresh.shape)

# make bottom 2 rows black where they are white the full width of the image
thresh[hh-3:hh, 0:ww] = 0

# # get bounds of white pixels
white = np.where(thresh==255)
xmin, ymin, xmax, ymax = np.min(white[1]), np.min(white[0]), np.max(white[1]), np.max(white[0])
print(xmin,xmax,ymin,ymax)

# crop the image at the bounds adding back the two blackened rows at the bottom
crop = img[ymin:ymax, xmin:xmax]

掩码尝试代码:

def select_descriptor_methods(image, method=None):    
    
    assert method is not None, "Please define a feature descriptor method. accepted Values are: 'sift', 'surf', 'brisk', 'orb'"

    if method == 'sift':
        descriptor = cv2.SIFT_create()
    elif method == 'surf':
        descriptor = cv2.xfeatures2d.SURF_create()
    elif method == 'brisk':
        descriptor = cv2.BRISK_create()
    elif method == 'orb':
        descriptor = cv2.ORB_create()

    # I tried to add this for masking
    _, mask = cv2.threshold(image, thresh=0, maxval=255, type=cv2.THRESH_BINARY)
    
    # and add the mask to this code (previously its "detectAndCompute(image, None)")
    (keypoints, features) = descriptor.detectAndCompute(image, mask)
    
    return (keypoints, features)
问题排查与分析

1. 掩码实现错误

当前掩码生成逻辑完全不符合需求:

_, mask = cv2.threshold(image, thresh=0, maxval=255, type=cv2.THRESH_BINARY)
  • THRESH_BINARY会把所有灰度值>0的像素设为255(保留),0设为0(屏蔽),这和你要“忽略黑背景”的逻辑相反,应该屏蔽的是纯黑区域,保留有效图像
  • 阈值设为0会把接近黑的暗部细节也误判为背景,导致有效特征丢失

2. 单应性矩阵方向与拼接逻辑缺陷

你的拼接逻辑是固定将查询图覆盖在画布左上角,只适用于查询图在基底图右侧/底部的场景:

result = cv2.warpPerspective(train_photo, Homography_Matrix,  (width, height))
result[0:query_photo.shape[0], 0:query_photo.shape[1]] = query_photo
  • cv2.findHomography计算的是训练图到查询图的单应性,但拼接时需要的是将查询图映射到训练图的坐标系,方向搞反会导致变换后的图像位置完全错误
  • 固定大小的画布和覆盖位置,无法适配查询图在基底图顶部/左侧的情况

3. 黑背景累积干扰特征匹配

每次拼接后的图像包含大量黑背景,即便裁剪了有效区域,如果后续拼接时没有将裁剪后的图像作为新基底,黑背景会持续累积,干扰特征提取和匹配。

4. 特征匹配鲁棒性不足

  • RANSAC重投影阈值(reprojThresh=4)可能过小,导致正确的匹配点被过滤
  • SURF算法存在专利限制,部分环境下兼容性差,且对低纹理区域的特征提取效果不稳定
修复建议

1. 修正掩码生成逻辑

生成仅保留非纯黑区域的掩码,避免误屏蔽有效图像:

def select_descriptor_methods(image, method=None):    
    assert method is not None, "Please define a feature descriptor method. accepted Values are: 'sift', 'surf', 'brisk', 'orb'"

    if method == 'sift':
        descriptor = cv2.SIFT_create()
    elif method == 'surf':
        descriptor = cv2.xfeatures2d.SURF_create()
    elif method == 'brisk':
        descriptor = cv2.BRISK_create()
    elif method == 'orb':
        descriptor = cv2.ORB_create()

    # 生成掩码:保留灰度值>5的区域(避免误判暗部细节)
    _, mask = cv2.threshold(image, thresh=5, maxval=255, type=cv2.THRESH_BINARY)
    (keypoints, features) = descriptor.detectAndCompute(image, mask)
    
    return (keypoints, features)

2. 调整单应性矩阵与拼接逻辑

改为计算查询图到训练图的单应性,自动适配任意方向的拼接:

def homography_stitching(keypoints_train_img, keypoints_query_img, matches, reprojThresh):
    keypoints_train_img = np.float32([keypoint.pt for keypoint in keypoints_train_img])
    keypoints_query_img = np.float32([keypoint.pt for keypoint in keypoints_query_img])
    
    if len(matches) > 4:
        # 交换点集,计算查询图到训练图的单应性
        points_query = np.float32([keypoints_query_img[m.trainIdx] for m in matches])
        points_train = np.float32([keypoints_train_img[m.queryIdx] for m in matches])
        
        (H, status) = cv2.findHomography(points_query, points_train, cv2.RANSAC, reprojThresh)
        return (matches, H, status)
    else:
        return None

# 拼接时自动计算画布大小
if M is not None:
    matches, Homography_Matrix, status = M
    h_query, w_query = query_photo.shape[:2]
    # 计算查询图变换后的边界
    corners_query = np.float32([[0,0], [w_query,0], [w_query,h_query], [0,h_query]]).reshape(-1,1,2)
    corners_transformed = cv2.perspectiveTransform(corners_query, Homography_Matrix)
    
    # 合并训练图和变换后查询图的边界,确定最终画布尺寸
    all_corners = np.vstack((corners_transformed, 
                            np.float32([[0,0], [train_photo.shape[1],0], 
                                        [train_photo.shape[1],train_photo.shape[0]], [0,train_photo.shape[0]]]).reshape(-1,1,2)))
    x_min, y_min = np.int32(all_corners.min(axis=0).ravel() - 0.5)
    x_max, y_max = np.int32(all_corners.max(axis=0).ravel() + 0.5)
    
    # 生成平移矩阵,避免变换后出现负坐标
    translation = np.array([[1,0,-x_min], [0,1,-y_min], [0,0,1]])
    H_final = translation @ Homography_Matrix
    
    # 拼接图像
    result = cv2.warpPerspective(query_photo, H_final, (x_max - x_min, y_max - y_min))
    result[-y_min:-y_min+train_photo.shape[0], -x_min:-x_min+train_photo.shape[1]] = train_photo
    
    # 裁剪黑背景
    gray = cv2.cvtColor(result, cv2.COLOR_RGB2GRAY)
    _, thresh = cv2.threshold(gray, 5, 255, cv2.THRESH_BINARY)
    white = np.where(thresh == 255)
    xmin, ymin, xmax, ymax = np.min(white[1]), np.min(white[0]), np.max(white[1]), np.max(white[0])
    crop = result[ymin:ymax, xmin:xmax]

3. 优化特征匹配鲁棒性

  • 调整RANSAC阈值为reprojThresh=6,适配无人机图像的分辨率
  • 替换为SIFT或ORB算法,避免SURF的专利问题,提升特征提取稳定性
  • 增加匹配点筛选,比如只保留前200个最匹配的点:
matches = sorted(matches, key=lambda x: x.distance)[:200]

内容的提问来源于stack exchange,提问作者Delan

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.20 18:42:05