You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

NEON版本裁剪区域匹配代码性能优化求助:性能不及常规版本

NEON版本点-in-矩形检测性能优化方案

我编写了用于查找屏幕首个包含指定点的裁剪区域的测试代码,包含point_in_neon(NEON实现)和point_in(常规实现)两个函数,二者功能一致:遍历矩形列表,返回首个包含目标点的矩形索引,无匹配则返回-1。原本预期NEON版本性能更优,但实际运行速度慢于常规版本。

使用的编译命令:

${CC} -O2 -ftree-vectorize -o vcomp vcomp.c

核心性能瓶颈与优化方案

1. 预处理避免重复计算

原NEON函数每次调用都重复计算矩形的right和bottom值,这部分可以提前预处理到连续数组中,消除运行时冗余运算。

2. 优化内存加载方式

原代码用vld1q_lane_s32逐个加载矩形坐标,效率远低于批量加载。将矩形的x、y、right、bottom分别打包成连续数组,用vld1q_s32一次性加载4个值,充分利用NEON的向量带宽。

3. 减少内存回写开销

原代码将NEON寄存器结果存回内存再判断匹配项,可直接通过寄存器操作查找第一个匹配的索引,避免不必要的内存读写。

4. 编译参数针对性优化

添加NEON专用编译参数,确保编译器生成最优的NEON指令:-mfpu=neon-vfpv4 -mfloat-abi=hard


优化后的完整代码

#include <stdio.h>
#include <stdlib.h>
#include <stdint.h>
#include <string.h>
#include <assert.h>
#include <math.h>
#include <sys/time.h>
#include <time.h>
#include <arm_neon.h>

#define WIDTH   (4096)
#define HEIGHT  (4096)
#define CLIPS   (32)
#define CLIP_QUADS (CLIPS / 4)

static inline uint64_t now(void) {
    struct timeval tv;
    gettimeofday(&tv,NULL);
    return tv.tv_sec*1000000+tv.tv_usec;
}

typedef struct _rect_t {
    int32_t x;
    int32_t y;
    uint32_t width;
    uint32_t height;
} rect_t;

typedef struct _point_t {
    int32_t x;
    int32_t y;
} point_t;

// 预处理矩形数据,打包成适合NEON批量处理的格式
typedef struct _rect_neon_pack_t {
    int32_t x[CLIPS];
    int32_t y[CLIPS];
    int32_t right[CLIPS];
    int32_t bottom[CLIPS];
} rect_neon_pack_t;

static void preprocess_rects(const rect_t* rs, rect_neon_pack_t* pack) {
    for (int i = 0; i < CLIPS; i++) {
        pack->x[i] = rs[i].x;
        pack->y[i] = rs[i].y;
        pack->right[i] = rs[i].x + rs[i].width - 1;
        pack->bottom[i] = rs[i].y + rs[i].height - 1;
    }
}

int32_t inline point_in_neon(const point_t *pt, const rect_neon_pack_t* pack, int quad_idx) {
    const int base = quad_idx * 4;
    int32x4_t px = vld1q_dup_s32(&pt->x);
    int32x4_t py = vld1q_dup_s32(&pt->y);

    // 批量加载4个矩形的x、right、y、bottom
    int32x4_t rx = vld1q_s32(&pack->x[base]);
    int32x4_t rright = vld1q_s32(&pack->right[base]);
    int32x4_t ry = vld1q_s32(&pack->y[base]);
    int32x4_t rbottom = vld1q_s32(&pack->bottom[base]);

    // 计算x方向匹配:(px >= rx) && (px <= rright)
    uint32x4_t x_match = vandq_u32(vcgeq_s32(px, rx), vcgeq_s32(rright, px));
    // 计算y方向匹配:(py >= ry) && (py <= rbottom)
    uint32x4_t y_match = vandq_u32(vcgeq_s32(py, ry), vcgeq_s32(rbottom, py));
    // 合并匹配结果
    uint32x4_t match = vandq_u32(x_match, y_match);

    // 查找第一个匹配的索引,无需存回内存
    uint32_t match_mask = vgetq_lane_u32(match, 0) | (vgetq_lane_u32(match, 1) << 1) | 
                          (vgetq_lane_u32(match, 2) << 2) | (vgetq_lane_u32(match, 3) << 3);
    if (match_mask == 0) return -1;
    // 找最低位的1,对应第一个匹配的索引
    return __builtin_ctz(match_mask);
}

int32_t inline point_in(const point_t *pt, const rect_t *rs, uint32_t len) {
    for(int32_t i=0;i<len;i++) {
        int32_t right=rs[i].x+rs[i].width-1,
                bottom=rs[i].y+rs[i].height-1;
                
        if(pt->x>=rs[i].x && pt->x<=right &&
           pt->y>=rs[i].y && pt->y<=bottom)
            return i;
    }
    return -1;
}

int32_t main(int32_t argc, char *argv[]) {
    rect_t rs[CLIPS];
    rect_neon_pack_t pack;
    
    int32_t i, j;
    uint64_t ts0, ts1;
    int32_t res[2][CLIPS];
    
    srand((unsigned int)time(NULL));
    for(i=0;i<CLIPS;i++) {
        rs[i].x=rand()%WIDTH;
        rs[i].y=rand()%HEIGHT;
        rs[i].width=rand()%WIDTH;
        rs[i].height=rand()%HEIGHT;
    }
    preprocess_rects(rs, &pack);
    memset(res, 0, sizeof(res));

    ts0=now();
    for(i=0;i<HEIGHT;i++) {
        for(j=0;j<WIDTH;j++) {
            point_t p={i, j};
            int32_t idx=point_in(&p, rs, CLIPS);
            if(idx>=0)
                res[0][idx]=1;
        }
    }
    ts0=now()-ts0;

    ts1=now();
    for(i=0;i<HEIGHT;i++) {
        for(j=0;j<WIDTH;j++) {
            int32_t k, idx = -1;
            point_t p={i, j};
            for(k=0;k<CLIP_QUADS;k++) {
                idx=point_in_neon(&p, &pack, k);
                if(idx>=0) {
                    idx = k*4 + idx;
                    break;
                }
            }
            if(idx>=0)
                res[1][idx]=1;
        }
    }
    ts1=now()-ts1;

    // 验证结果一致性
    for(i=0;i<CLIPS;i++) {
        if(res[0][i]!=res[1][i]) {
            printf("error at index %d\n", i);
            return 1;
        }
    }

    printf("regular = %lu us\n", ts0);
    printf("neon    = %lu us\n", ts1);
   
    return 0;
}

优化后编译命令

${CC} -O2 -ftree-vectorize -mfpu=neon-vfpv4 -mfloat-abi=hard -o vcomp vcomp.c

内容的提问来源于stack exchange,提问作者Bruce Hsu

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.06 11:30:57