You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Jetson AGX Xavier内核与用户态共享内存缓存异常问题

内核态与用户态共享内存缓存一致性问题(Jetson AGX Xavier)

实现背景与流程

基于NVIDIA Jetson AGX Xavier(ARM架构)开发内核模块,通过以下流程创建内核态与用户态共享内存:

  • 内核态调用kzalloc()分配内存
  • 调用dma_map_single()获取该内存的总线地址
  • 通过remap_pfn_range()将内存映射到用户态
  • 用户态调用mmap()获取共享内存指针,使用memcpy()写入数据
  • 内核态读取共享内存,验证是否为用户态写入的最新数据

问题现象

内核态读取共享内存时,数据存在一致性问题:有时能读到最新数据,有时不能。怀疑是缓存相关问题,已在内核模块中使用pgprot_noncached()将内存设置为非缓存,但问题仍未解决。

内核模块关键代码

static int tx1_mem_get(struct file* file_ptr, struct vm_area_struct* mem_struct)
{
    int ret_val;
    unsigned int i;
    unsigned long start_addr;
    unsigned long page_frame_num;
    unsigned long mem_size;
    struct device* device_ptr;

    if (tx1_mem_count < BUFFER_COUNT) {
        /* Obtain the device pointer for the device which is requesting the memory */
        device_ptr = &pcidev_global_ptr->dev;

        /* Get the array index to store the addresses */
        i = tx1_mem_count;

        /* Allocate TX1 Buffer */
        tx1_data_buffers[i] = kzalloc(BUFFER_SIZE, GFP_DMA | GFP_ATOMIC);
        if (tx1_data_buffers[i] == NULL) {
            sprintf(msg_buffer, "WARNING: failed to allocate tx1 buffer %d", i);
            umtrx_ep_msg(msg_buffer);
            return 0;
        }

        /* Get the TX1 DMA address */
        tx1_dma_buffers[i] = dma_map_single(device_ptr, tx1_data_buffers[i], BUFFER_SIZE, DMA_BIDIRECTIONAL);
        if (tx1_dma_buffers[i] == DMA_MAPPING_ERROR) {
            sprintf(msg_buffer, "WARNING: failed to obtain dma memory for tx1 buffer %d", i);
            umtrx_ep_msg(msg_buffer);
            kfree(tx1_data_buffers[i]);
            tx1_data_buffers[i] = NULL;
            return 0;
        }

        /* Set memory attribute flags (VM_DONTEXPAND | VM_DONTDUMP = VM_RESERVED; VM_RESERVED flag not supported in new kernel versions)*/
        mem_struct->vm_flags |= VM_READ | VM_WRITE | VM_SHARED | VM_LOCKED | VM_DONTEXPAND | VM_DONTDUMP;

        /* Set the page to non-cached */
        mem_struct->vm_page_prot = pgprot_noncached(mem_struct->vm_page_prot);

        /* Obtain required parameters for creating a mapping */
        start_addr = mem_struct->vm_start;
        page_frame_num = virt_to_phys(tx1_data_buffers[i]) >> PAGE_SHIFT;
        mem_size = mem_struct->vm_end - mem_struct->vm_start;
        if (mem_size > BUFFER_SIZE) {
            sprintf(msg_buffer, "WARNING: couldn't map tx1 buffer %d, can't map more than %d bytes", i, BUFFER_SIZE);
            umtrx_ep_msg(msg_buffer);
            dma_unmap_single(device_ptr, tx1_dma_buffers[i], BUFFER_SIZE, DMA_BIDIRECTIONAL);
            tx1_dma_buffers[i] = 0;
            kfree(tx1_data_buffers[i]);
            tx1_data_buffers[i] = NULL;
            return 0;
        }

        ret_val = remap_pfn_range(mem_struct, start_addr, page_frame_num, mem_size, mem_struct->vm_page_prot);
        if (ret_val != 0) {
            sprintf(msg_buffer, "WARNING: couldn't map tx1 buffer %d", i);
            umtrx_ep_msg(msg_buffer);
            dma_unmap_single(device_ptr, tx1_dma_buffers[i], BUFFER_SIZE, DMA_BIDIRECTIONAL);
            tx1_dma_buffers[i] = 0;
            kfree(tx1_data_buffers[i]);
            tx1_data_buffers[i] = NULL;
            return 0;
        }

        /* Reserve the obtained memory, so that it is not swapped out by the kernel. Kernel space and user space can both access this buffer, so it must be
         * reserved. */
        reserve_mem_buffer(tx1_data_buffers[i], (unsigned int)(BUFFER_SIZE));

        /* Memory has been obtained, and reserved. Now populate it with the physical address using GPC-DMA */
        write_phys_addr(tx1_data_buffers[i]);

        sprintf(msg_buffer, "tx1 buffer %d acquired", i);
        umtrx_ep_msg(msg_buffer);

        tx1_mem_count = tx1_mem_count + 1;
    } else {
        umtrx_ep_msg("WARNING: failed to acquire tx1 buffer. tx1 memory is full");
    }

    return 0;
}

已尝试的解决方法

  • 使用dma_mmap_attrs()替代remap_pfn_range(),问题无改善
  • 禁用SMMU并在用户态调用显式缓存刷新代码,问题可解决,但因应用限制无法禁用SMMU。用户态缓存刷新代码如下:
void flush_cache(void* start, size_t size)
{
    // Ensure start address is aligned to cache line size
    uintptr_t addr = (uintptr_t)start & ~(uintptr_t)(CACHE_LINE_SIZE - 1);

    // Calculate the end address aligned to cache line size
    uintptr_t end = ((uintptr_t)start + size + CACHE_LINE_SIZE - 1) & ~(uintptr_t)(CACHE_LINE_SIZE - 1);

    // Flush the cache for each cache line in the specified range
    for (; addr < end; addr+= CACHE_LINE_SIZE) {
        __asm__ __volatile__("dc civac, %0" : : "r"(addr));
    }

    // Ensure completion of cache maintenance operations
    __asm__ __volatile("dsb sy");
}

疑问

按理解,当前的共享内存创建方式应生成非缓存内存,但缓存一致性问题依然存在,请问哪里操作有误?

注:共享内存大小为4MB,尝试缩小至4KB(内核kzalloc可分配的最小页大小)后问题仍存在。

内容的提问来源于stack exchange,提问作者Sana Ur Rehman

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.28 21:57:47