You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用cuSPARSE的cusparseCsr2cscEx2()执行矩阵转置时遇内部错误

Fixing "Internal Error" in cuSPARSE cusparseCsr2cscEx2 Matrix Transpose

Let's cut to the chase: your "internal error" stems from a simple but critical mistake—you're passing host-side (CPU) array pointers to cusparseCsr2cscEx2 and its buffer size calculation function, but these APIs require device-side (GPU) memory pointers for input data. When cuSPARSE tries to access GPU memory using invalid host addresses, it throws that vague internal error.

Breakdown of the Fix

  • The input parameters for cusparseCsr2cscEx2_bufferSize and cusparseCsr2cscEx2 must point to memory allocated on the GPU. You already copied your host data to device pointers (d_csr_offsets, d_csr_columns, d_csr_values)—use those instead of the h_* host arrays.
  • Your initial cusparseCreate call wasn't using your CHECK_CUSPARSE macro, which could miss initialization issues. Let's fix that too.
  • Don't forget to clean up device memory and the cuSPARSE handle at the end to avoid leaks.

Corrected Full Code

#include <cuda_runtime_api.h> // cudaMalloc, cudaMemcpy, etc.
#include <cusparse.h> // cusparseSparseToDense
#include <stdio.h> // printf
#include <stdlib.h> // EXIT_FAILURE

#define CHECK_CUDA(func) \
{ \
    cudaError_t status = (func); \
    if (status != cudaSuccess) { \
        printf("CUDA API failed at line %d with error: %s (%d)\n", \
               __LINE__, cudaGetErrorString(status), status); \
        return EXIT_FAILURE; \
    } \
}

#define CHECK_CUSPARSE(func) \
{ \
    cusparseStatus_t status = (func); \
    if (status != CUSPARSE_STATUS_SUCCESS) { \
        printf("CUSPARSE API failed at line %d with error: %s (%d)\n", \
               __LINE__, cusparseGetErrorString(status), status); \
        return EXIT_FAILURE; \
    } \
}

int main(void) {
    // CUSPARSE APIs
    cusparseHandle_t handle = NULL;
    CHECK_CUSPARSE(cusparseCreate(&handle))

    // Initialize matrix A
    int num_rows = 5;
    int num_cols = 4;
    int nnz = 11;
    int h_csr_offsets[] = { 0, 3, 4, 7, 9, 11 };
    int h_csr_columns[] = { 0, 2, 3, 1, 0, 2, 3, 1, 3, 1, 2 };
    float h_csr_values[] = { 1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 6.0f, 7.0f, 8.0f, 9.0f, 10.0f, 11.0f };

    // Device memory management
    int* d_csr_offsets, * d_csr_columns;
    float* d_csr_values;
    CHECK_CUDA(cudaMalloc((void**)&d_csr_offsets, (num_rows + 1) * sizeof(int)))
    CHECK_CUDA(cudaMalloc((void**)&d_csr_columns, nnz * sizeof(int)))
    CHECK_CUDA(cudaMalloc((void**)&d_csr_values, nnz * sizeof(float)))
    CHECK_CUDA(cudaMemcpy(d_csr_offsets, h_csr_offsets, (num_rows + 1) * sizeof(int), cudaMemcpyHostToDevice))
    CHECK_CUDA(cudaMemcpy(d_csr_columns, h_csr_columns, nnz * sizeof(int), cudaMemcpyHostToDevice))
    CHECK_CUDA(cudaMemcpy(d_csr_values, h_csr_values, nnz * sizeof(float), cudaMemcpyHostToDevice))

    // Memory allocation of transpose A
    int* d_csr_offsets_AT, * d_csr_columns_AT;
    float* d_csr_values_AT;
    CHECK_CUDA(cudaMalloc((void**)&d_csr_offsets_AT, (num_cols + 1) * sizeof(int)))
    CHECK_CUDA(cudaMalloc((void**)&d_csr_columns_AT, nnz * sizeof(int)))
    CHECK_CUDA(cudaMalloc((void**)&d_csr_values_AT, nnz * sizeof(float)))

    size_t buffer_temp_size;
    // Use device pointers for input here
    CHECK_CUSPARSE(cusparseCsr2cscEx2_bufferSize(
        handle, num_rows, num_cols, nnz,
        d_csr_values, d_csr_offsets, d_csr_columns,
        d_csr_values_AT, d_csr_offsets_AT, d_csr_columns_AT,
        CUDA_R_32F, CUSPARSE_ACTION_NUMERIC, CUSPARSE_INDEX_BASE_ZERO, CUSPARSE_CSR2CSC_ALG1,
        &buffer_temp_size))
    void* buffer_temp = NULL;
    printf("buffer_temp_size is %zd\n", buffer_temp_size);
    CHECK_CUDA(cudaMalloc(&buffer_temp, buffer_temp_size))

    // And use device pointers for input here too
    CHECK_CUSPARSE(cusparseCsr2cscEx2(handle, num_rows, num_cols, nnz,
                                      d_csr_values, d_csr_offsets, d_csr_columns,
                                      d_csr_values_AT, d_csr_offsets_AT, d_csr_columns_AT,
                                      CUDA_R_32F, CUSPARSE_ACTION_NUMERIC, CUSPARSE_INDEX_BASE_ZERO, CUSPARSE_CSR2CSC_ALG1,
                                      buffer_temp))

    // Cleanup
    CHECK_CUDA(cudaFree(buffer_temp))
    CHECK_CUDA(cudaFree(d_csr_values_AT))
    CHECK_CUDA(cudaFree(d_csr_columns_AT))
    CHECK_CUDA(cudaFree(d_csr_offsets_AT))
    CHECK_CUDA(cudaFree(d_csr_values))
    CHECK_CUDA(cudaFree(d_csr_columns))
    CHECK_CUDA(cudaFree(d_csr_offsets))
    CHECK_CUSPARSE(cusparseDestroy(handle))

    printf("Matrix transpose completed successfully!\n");
    return EXIT_SUCCESS;
}

Additional Tips

  • Always verify that input pointers to cuSPARSE APIs are device-side unless the documentation explicitly states otherwise.
  • Your setup (GTX 1080, CUDA 11.1, Windows 10) is fully compatible—Pascal GPUs are supported in CUDA 11.x.
  • Keep using those error-checking macros—they're invaluable for catching issues like this before they turn into vague internal errors.

内容的提问来源于stack exchange,提问作者hellohawaii

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.11 07:51:00