You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于Numba的CUDA机器人正运动学实现性能异常低下问题

机器人正运动学计算的CUDA性能优化

我正尝试优化机器人领域中随机正运动学方程的性能。该计算函数接收6个关节值作为输入,输出一个表示机器人末端执行器位姿的4×4数组。数组的每个元素可独立计算,涉及大量对6个输入值的乘法、正弦及余弦运算。扩展到N个配置时,核心函数接收(N,6)输入数组并返回(N,4,4)数组。

性能测试结果

我的目标是对比不同实现的性能,找出用Numba加速正运动学计算的性能极限。以下是针对2,000,000×6数组的执行时间:

  • @numba.guvectorize(target=cpu/parallel/cuda)
    • cpu:92ms
    • parallel:21ms
    • cuda:200ms(仅处理时间)
  • @cuda.jit:160ms(仅处理时间)

CUDA实现优化需求

我寻求优化正运动学计算的CUDA实现的建议:

  • 采用target="cuda"的@guvectorize:我认为该实现已基于最小输入维度优化,最大化了CUDA的并行度,但不确定是否还有优化空间。
  • @cuda.jit:作为CUDA编程新手,我认为此实现仍有优化空间,已尝试的方案包括:
    • 启动N个块,每个块16线程,用条件语句分配线程计算4×4数组的特定元素,但知道条件语句不符合CUDA范式,效率不高。
    • 启动独立核函数计算4×4数组的每个元素,可消除条件语句,但可能因大量核函数启动带来显著开销。

疑问

  1. 此类计算是否适合GPU?虽然计算密集,但4×4输出数组的每个元素需要完全不同的指令。
  2. 采用target="cuda"的@guvectorize实现是否还有提速空间?
  3. 对于@cuda.jit实现,有经验的开发者通常会采用什么方案处理此类计算?

可运行代码

以下是可运行代码(需安装Numba和CUDA):

from time import perf_counter

import numpy as np
from numba import bool_, cuda, float64, guvectorize


@guvectorize(
    [(float64[:], bool_[:, :], float64[:, :])],
    "(n), (m,m) -> (m,m)",
    nopython=True,
    target="cuda",
)
def guvectorized(joints, dummy, res):
    q1, q2, q3, q4, q5, q6 = joints
    d1, a1, a2, a3, d4, d6 = float64(10.), float64(2.), float64(3.), float64(3.), float64(2.), float64(5.)
    c1, c2, c4, c5, c6 = np.cos(q1), np.cos(q2), np.cos(q4), np.cos(q5), np.cos(q6)
    s1, s2, s4, s5, s6 = np.sin(q1), np.sin(q2), np.sin(q4), np.sin(q5), np.sin(q6)
    s23, c23 = np.sin(q3 + q2), np.cos(q2 + q3)

    res[0,0] = c6 * (c5 * (c1 * c23 * c4 + s1 * s4) - c1 * s23 * s5) + s6 * (
        s1 * c4 - c1 * c23 * s4
    )
    res[1, 0] = c6 * (c5 * (s1 * c23 * c4 + c1 * s4) - s1 * s23 * s5) - s6 * (
        c1 * c4 - s1 * c23 * s4
    )
    res[2,0] = c6 * (c23 * s5 + s23 * c4 * c5) - s23 * s4 * s6
    res[0,1] = s6 * (
        c1 * s23 * s5 - c5 * (s1 * c23 * c4 + s1 * s4) + c6 * (s1 * c4 - c1 * c23 * s4)
    )
    res[1,1] = s6 * (s1 * s23 * s5 - c5 * (s1 * c23 * c4 + c1 * s4)) - c6 * (
        c1 * c4 + s1 * c23 * s4
    )
    res[2,1] = -s6 * (c23 * s5 + s23 * c4 * c5) - s23 * s4 * c6
    res[0,2] = s5 * (c1 * c23 * c4 + s1 * s4) + c1 * s23 * c5
    res[1,2] = s5 * (s1 * c23 * c4 - c1 * s4) + s1 * s23 * c5
    res[2,2] = s23 * c4 * c5 - c23 * c5
    res[0,3] = d6 * (s5 * (c1 * c23 * c4 + s1 + s4) + c1 * s23 * c5) + c1 * (
        a1 + a2 * c2 + a3 * c23 + d4 * s23
    )
    res[1,3] = d6 * (s5 * (s1 * c23 * c4 + c1 * s4) + s1 * s23 * c5) + s1 * (
        a1 + a2 * c2 + a3 * c23 + d4 * s23
    )
    res[2,3] = a2 * s2 + d1 + a3 * s23 - d4 * c23 + d6 * (s23 * c4 * s5 - c23 * c5)
    res[3, 0] = 0.0
    res[3, 1] = 0.0
    res[3, 2] = 0.0
    res[3, 3] = 1.0

@cuda.jit
def cuda_jit(joints, res):
    i = cuda.threadIdx.x
    block_id = cuda.blockIdx.x
    q1, q2, q3, q4, q5, q6 = joints[block_id]
    d1, a1, a2, a3, d4, d6 = float64(10.), float64(2.), float64(3.), float64(3.), float64(2.), float64(5.)
    c1, c2, c4, c5, c6 = np.cos(q1), np.cos(q2), np.cos(q4), np.cos(q5), np.cos(q6)
    s1, s2, s4, s5, s6 = np.sin(q1), np.sin(q2), np.sin(q4), np.sin(q5), np.sin(q6)
    s23, c23 = np.sin(q3 + q2), np.cos(q2 + q3)


    if i == 0:
        res[block_id, 0, 0] = c6 * (c5 * (c1 * c23 * c4 + s1 * s4) - c1 * s23 * s5) + s6 * (
            s1 * c4 - c1 * c23 * s4
        )
    elif i == 1:
        res[block_id, 1, 0] = c6 * (c5 * (s1 * c23 * c4 + c1 * s4) - s1 * s23 * s5) - s6 * (
            c1 * c4 - s1 * c23 * s4
        )
    elif i == 2:
        res[block_id, 2, 0] = c6 * (c23 * s5 + s23 * c4 * c5) - s23 * s4 * s6
    elif i == 3:
        res[block_id, 0, 1] = s6 * (
            c1 * s23 * s5 - c5 * (s1 * c23 * c4 + s1 * s4) + c6 * (s1 * c4 - c1 * c23 * s4)
        )
    elif i == 4:
        res[block_id, 1, 1] = s6 * (s1 * s23 * s5 - c5 * (s1 * c23 * c4 + c1 * s4)) - c6 * (
            c1 * c4 + s1 * c23 * s4
        )
    elif i == 5:
        res[block_id, 2, 1] = -s6 * (c23 * s5 + s23 * c4 * c5) - s23 * s4 * c6
    elif i == 6:
        res[block_id, 0, 2] = s5 * (c1 * c23 * c4 + s1 * s4) + c1 * s23 * c5
    elif i == 7:
        res[block_id, 1, 2] = s5 * (s1 * c23 * c4 - c1 * s4) + s1 * s23 * c5
    elif i == 8:
        res[block_id, 2, 2] = s23 * c4 * c5 - c23 * c5
    elif i == 9:
        res[block_id, 0, 3] = d6 * (s5 * (c1 * c23 * c4 + s1 + s4) + c1 * s23 * c5) + c1 * (
            a1 + a2 * c2 + a3 * c23 + d4 * s23
        )
    elif i == 10:
        res[block_id, 1, 3] = d6 * (s5 * (s1 * c23 * c4 + c1 * s4) + s1 * s23 * c5) + s1 * (
            a1 + a2 * c2 + a3 * c23 + d4 * s23
        )
    elif i == 11:
        res[block_id, 2, 3] = a2 * s2 + d1 + a3 * s23 - d4 * c23 + d6 * (s23 * c4 * s5 - c23 * c5)
    elif i == 12:
        res[block_id, 3, 0] = float64(0.0)
        res[block_id, 3, 1] = float64(0.0)
        res[block_id, 3, 2] = float64(0.0)
        res[block_id, 3, 3] = float64(1.0)


if __name__ == "__main__":
    # Setup
    N = 1_000_000
    # Guvectorized arrays
    input_arr = np.random.rand(N, 6)
    dummy_arr = np.empty((4, 4), dtype=np.bool_)
    out = guvectorized(input_arr, dummy_arr)

    # Cuda.jit arrays
    input_arr_cu = cuda.to_device(input_arr)
    output_arr_cu = cuda.device_array((N, 4, 4))
    threads_per_block = 13
    blocks_per_grid = N
    cuda_jit[blocks_per_grid, threads_per_block](input_arr_cu, output_arr_cu)

    # Timing
    #   Guvectorize
    init = perf_counter()
    guvectorized(input_arr, dummy_arr)
    print(f"Time `guvectorize`: {perf_counter() - init}")

    #   Cuda jit
    init = perf_counter()
    cuda_jit[blocks_per_grid, threads_per_block](input_arr_cu, output_arr_cu)
    cuda.synchronize()
    print(f"Time `cuda.jit`: {perf_counter() - init}")

内容的提问来源于stack exchange,提问作者Manu

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.19 01:57:02